From 391b3a473294afe427a1ca836c3bfad2c444ae0f Mon Sep 17 00:00:00 2001 From: wuyang <5700876+banisherwy@user.noreply.gitee.com> Date: Wed, 8 Jul 2026 12:25:30 +0800 Subject: [PATCH] Expand arXiv paper corpus --- .../2026-07-08-arxiv-expanded-sweep.md | 55 + ...t-candidates-2025-07-08-to-2026-07-08.json | 53788 ++++++++++++++++ ...agent-papers-2025-07-08-to-2026-07-08.json | 35331 ++++++++++ data/index.json | 39106 +++++++++++ data/summary.json | 106 +- papers/README.md | 1 + papers/corpus-summary-2026-07-08.md | 143 + ...tial-decision-making-in-language-models.md | 61 + ...-calling-for-relative-evaluation-of-ano.md | 63 + ...-context-protocol-mcp-tool-use-benchmar.md | 60 + ...agentic-recovery-from-external-failures.md | 61 + ...tion-for-agentic-multi-turn-interaction.md | 61 + ...-for-automating-distribution-grid-analy.md | 63 + ...long-horizon-and-efficient-mobile-agent.md | 66 + ...or-power-system-analysis-and-operations.md | 60 + ...ture-for-geospatial-analysis-function-c.md | 62 + ...aluate-agent-architectures-in-enterpris.md | 64 + ...ic-intelligence-via-environment-scaling.md | 59 + ...ngual-and-regionalized-agent-evaluation.md | 60 + ...uation-of-llm-agents-beyond-final-state.md | 61 + ...tion-free-controllable-evaluation-frame.md | 61 + ...-a-survey-of-architectures-capabilities.md | 65 + ...rning-with-a-multi-turn-multi-task-fram.md | 59 + ...eyond-utility-an-open-ended-perspective.md | 61 + ...ework-for-llm-based-multi-agent-applica.md | 61 + ...gal-behavior-of-llm-agents-under-eu-law.md | 61 + ...guage-model-susceptibility-in-agent-to-.md | 62 + ...ta-synthesis-solution-for-real-world-mu.md | 61 + ...owards-agentic-tool-use-reward-modeling.md | 60 + ...-llm-agents-via-environment-interaction.md | 63 + ...-through-multi-agent-interaction-in-vqa.md | 63 + ...pi-centric-llm-agent-defense-frameworks.md | 62 + ...all-language-models-for-agentic-tasks-o.md | 60 + ...r-general-ai-agents-a-technical-white-p.md | 65 + ...entic-reasoning-in-the-neurips-cure-ben.md | 65 + ...l-use-data-via-multi-agent-role-playing.md | 62 + ...earning-for-agentic-information-seeking.md | 65 + ...t-small-language-coder-model-mify-coder.md | 64 + ...ation-of-llm-agents-under-real-world-ap.md | 60 + ...and-execution-of-llm-generated-programs.md | 64 + ...ompt-caching-for-long-horizon-agentic-t.md | 61 + ...xt-engineering-for-agentic-data-science.md | 60 + ...del-paper-reading-agents-more-efficient.md | 62 + ...-multi-agent-reasoning-through-holistic.md | 63 + ...-deployable-in-real-world-dynamic-envir.md | 61 + ...-agent-memory-with-systematic-benchmark.md | 63 + ...gent-creation-for-agentic-orchestration.md | 63 + ...-language-agents-under-communication-ba.md | 61 + ...gents-in-realistic-negotiation-scenario.md | 61 + ...fficient-agent-defense-with-hierarchica.md | 62 + ...lutionary-security-evaluation-of-agents.md | 60 + ...erabilities-across-deep-research-agents.md | 64 + ...nder-controllable-and-extreme-context-g.md | 61 + ...d-detecting-tool-use-hallucinations-via.md | 62 + ...g-and-benchmarking-attacks-on-openclaw-.md | 63 + ...ramework-for-agent-system-observability.md | 62 + ...-agent-safety-through-incident-response.md | 61 + ...fending-multi-turn-safety-risks-in-tool.md | 65 + ...-with-episodic-memory-in-language-agent.md | 62 + ...lls-for-agentic-ai-through-hybrid-model.md | 59 + ...-framework-for-long-horizon-search-agen.md | 62 + ...adaptive-trust-calibration-in-model-con.md | 62 + ...ety-alignment-in-vision-language-agents.md | 59 + ...ond-single-channel-agentic-benchmarking.md | 61 + ...iation-as-a-causal-mechanism-of-agent-f.md | 62 + ...perception-vulnerability-in-llm-driven-.md | 63 + ...gents-with-parametric-reflective-memory.md | 61 + ...iminal-prompting-in-multi-agent-systems.md | 61 + ...ctured-analysis-and-reporting-of-agenti.md | 63 + ...ini-internets-for-diagnosing-epistemic-.md | 62 + ...ime-via-dynamic-importance-estimation-f.md | 64 + ...us-llm-fine-tuning-with-language-agents.md | 61 + ...pproach-to-study-affective-polarization.md | 60 + ...rnance-framework-for-military-ai-agents.md | 63 + ...uage-agents-toward-strategic-exploratio.md | 61 + ...tform-for-function-calling-data-synthes.md | 63 + ...benchmark-for-self-evolving-language-ag.md | 61 + ...rchical-autonomy-evolution-of-ai-agents.md | 64 + ...ion-of-data-over-exposure-in-llm-agents.md | 62 + ...-are-language-agents-from-human-experts.md | 63 + ...e-to-metal-kernel-generation-on-emergin.md | 62 + ...-considerations-for-multi-agent-systems.md | 66 + ...sis-with-evidence-integrated-language-a.md | 63 + ...pe-of-agentic-ai-a-comprehensive-survey.md | 61 + ...-through-multi-agent-dialectical-negoti.md | 61 + ...-for-tool-use-under-complex-constraints.md | 61 + ...more-precise-instructions-for-language-.md | 59 + ...zed-llm-agents-the-curious-case-of-ment.md | 62 + ...tomated-alzheimer-s-disease-detection-w.md | 61 + ...ion-and-coverage-audit-of-llm-agent-too.md | 63 + ...work-for-formalizing-llm-agent-security.md | 61 + ...a-geometry-aware-vision-language-agents.md | 62 + ...lay-for-llm-agent-trajectory-relabeling.md | 62 + ...hierarchical-memory-for-language-agents.md | 64 + ...or-persistent-and-semantically-consiste.md | 60 + ...system-for-autonomous-industrial-safety.md | 62 + ...-cross-the-line-before-they-actually-do.md | 60 + ...e-usage-of-agents-with-real-world-tools.md | 63 + ...eral-purpose-agent-for-open-agentic-web.md | 64 + ...-for-small-uas-separation-assurance-und.md | 64 + ...jectory-benchmark-for-safety-evaluation.md | 62 + ...ought-budget-effects-in-function-callin.md | 61 + ...and-internal-reward-for-language-agents.md | 63 + ...upled-latent-reasoning-for-agent-safety.md | 62 + ...lexity-for-tool-augmented-language-agen.md | 62 + ...t-emerging-supply-chain-injections-in-a.md | 60 + ...ulecon-agentic-security-rule-conversion.md | 60 + ...optimization-for-safe-multi-agent-navig.md | 65 + ...c-agentic-data-reactivates-general-tool.md | 59 + ...-user-instructions-expose-critical-vuln.md | 63 + ...tion-data-and-evaluation-for-llm-agents.md | 60 + ...why-ai-agents-that-think-must-never-act.md | 63 + ...ntic-llm-driven-obfuscation-for-ip-prot.md | 61 + ...ing-information-and-control-to-filesyst.md | 61 + ...trinsic-non-attack-trajectory-benchmark.md | 62 + ...odied-vision-language-agent-framework-f.md | 63 + ...do-harmful-skills-weaponize-your-agents.md | 62 + ...ty-symbolic-guardrails-for-domain-speci.md | 63 + ...ge-reliability-propagation-cascades-and.md | 61 + ...ection-architecture-for-agentic-systems.md | 65 + ...issing-threat-model-for-ai-agent-safety.md | 64 + ...hitectures-for-offensive-security-tasks.md | 62 + ...d-harm-recovery-for-computer-use-agents.md | 65 + ...imization-framework-for-language-agents.md | 62 + ...ot-be-it-mitigating-trust-boundary-conf.md | 62 + ...cks-novel-threats-for-function-calling-.md | 62 + ...f-vision-language-agents-for-spatial-re.md | 61 + ...context-fragmented-violations-in-multi-.md | 67 + ...pecifications-from-1-bit-danger-signals.md | 64 + ...nformation-flow-tracking-for-llm-agents.md | 62 + ...ecture-matters-for-multi-agent-security.md | 65 + ...-agents-with-efficient-dynamic-analysis.md | 63 + ...luation-of-ai-agent-security-guardrails.md | 61 + ...-for-open-source-llms-in-interactive-to.md | 59 + ...rk-for-automated-3d-cutscene-generation.md | 64 + ...idation-and-zero-trust-security-for-sem.md | 63 + ...l-firewall-for-structured-workflow-ai-a.md | 63 + ...rdrails-for-clinical-safety-hallucinati.md | 63 + ...ic-discovery-on-a-real-optical-platform.md | 63 + ...r-autonomous-agent-frameworks-a-layered.md | 63 + ...al-framework-for-proactive-embodied-age.md | 64 + ...ement-learning-in-large-language-models.md | 62 + ...y-efficient-red-teaming-for-prompt-inje.md | 62 + ...-contracts-for-agentic-security-systems.md | 63 + ...urity-pattern-selection-for-iot-systems.md | 62 + ...constraint-guided-large-language-agents.md | 63 + ...he-loop-ai-speech-therapy-agent-for-per.md | 68 + ...-framework-for-agent-safety-measurement.md | 61 + ...m-agents-in-real-world-ehr-environments.md | 63 + ...d-benchmark-rewriting-and-analogical-re.md | 64 + ...tration-for-small-language-model-agents.md | 63 + ...ork-for-pre-print-anomaly-detection-in-.md | 61 + ...or-root-cause-analysis-in-microservices.md | 61 + ...compilation-for-agentic-llm-deployments.md | 62 + ...ollable-and-interactive-red-teaming-pla.md | 64 + ...rieval-for-agentic-search-via-direct-co.md | 63 + ...ugmented-guardrail-for-llm-agent-safety.md | 63 + ...t-interference-in-llm-agent-scaffolding.md | 64 + ...arning-for-long-horizon-language-agents.md | 63 + ...yber-offense-forecast-consequences-and-.md | 64 + ...e-evolution-of-llm-agent-memory-mechani.md | 63 + ...or-reliable-llm-based-autonomous-agents.md | 64 + ...m-agents-a-unified-graph-representation.md | 64 + ...neral-sequential-decision-making-agents.md | 64 + ...interpretability-of-agentic-ai-tool-use.md | 63 + ...s-for-hierarchical-generalized-planning.md | 60 + ...-safety-fail-to-generalize-across-tasks.md | 60 + ...i-model-router-for-agentic-tool-calling.md | 61 + ...luating-llms-on-chemical-cost-reasoning.md | 62 + ...n-llm-agents-for-cyber-attack-scenarios.md | 60 + ...lving-memory-agents-over-provenance-dag.md | 64 + ...l-layers-a-mechanistic-evaluation-of-pe.md | 66 + ...ramework-for-automated-cyber-intrusions.md | 64 + ...-reasoning-level-denial-of-service-in-l.md | 60 + ...nd-accountability-from-models-to-agents.md | 63 + ...ention-verification-for-language-agents.md | 62 + ...dicts-action-control-in-language-agents.md | 62 + ...e-benchmark-for-evaluating-agent-values.md | 60 + ...-agentic-ai-systems-openclaw-case-study.md | 62 + ...s-of-llm-agents-in-real-os-environments.md | 62 + ...-a-rate-distortion-framework-for-agent-.md | 60 + ...y-argument-level-provenance-solves-enfo.md | 65 + ...in-llm-agents-via-trajectory-refinement.md | 64 + ...se-agents-via-structured-meta-cognition.md | 63 + ...tent-in-simulated-embodied-environments.md | 64 + ...marking-heterogeneous-geospatial-reason.md | 63 + ...ajectories-for-agentic-safety-alignment.md | 60 + ...ark-and-domain-randomized-rl-recipe-for.md | 60 + ...tual-trace-auditing-of-llm-agent-skills.md | 61 + ...fety-under-skill-facing-attack-surfaces.md | 68 + ...-engine-for-structure-aware-associative.md | 62 + ...ent-aware-structured-memory-for-long-ho.md | 62 + ...traversal-retrieval-with-planning-mecha.md | 62 + ...text-icu-data-a-benchmark-beyond-behavi.md | 66 + ...work-for-distributed-materials-informat.md | 64 + ...es-as-self-maintaining-software-ecosyst.md | 62 + ...gents-in-fast-healthcare-interoperabili.md | 63 + ...ld-adopt-the-plan-then-execute-paradigm.md | 65 + ...enchmark-for-real-world-teaching-workfl.md | 60 + ...guided-enforcement-for-llm-agent-memory.md | 62 + ...t-supply-chains-via-payload-less-skills.md | 62 + ...ent-memory-in-multi-party-conversations.md | 62 + ...learning-interatomic-potential-developm.md | 64 + ...collaboration-failure-attribution-and-s.md | 63 + ...-memory-in-large-vision-language-models.md | 62 + ...curing-ai-agents-like-operating-systems.md | 61 + ...-open-source-agentic-modeling-framework.md | 64 + ...n-framework-for-multimodal-agent-memory.md | 63 + ...arly-to-save-energy-in-consumer-devices.md | 65 + ...icle-monte-carlo-workflows-for-colloida.md | 64 + ...ng-and-retrieving-agent-memory-via-a-hy.md | 61 + ...rce-distributed-multimodal-agent-memory.md | 62 + ...ng-for-efficient-long-term-agent-memory.md | 61 + ...weight-updates-via-population-broadcast.md | 65 + ...g-video-understanding-via-online-indexi.md | 63 + ...ice-a-systematic-analysis-of-generator-.md | 63 + ...bustness-of-ai-enabled-security-orchest.md | 63 + ...-framework-for-resilient-multi-agent-ev.md | 61 + ...lm-agents-under-untrusted-tool-feedback.md | 61 + ...ture-for-long-horizon-scientific-agents.md | 64 + ...tric-memory-layer-for-software-reposito.md | 64 + ...esign-for-multi-agent-systems-in-port-h.md | 62 + ...mory-control-for-long-horizon-gui-agent.md | 63 + ...me-guarantee-architecture-is-structural.md | 60 + ...a-locally-correct-but-non-transferable-.md | 63 + ...s-for-efficient-and-accurate-llm-agents.md | 61 + ...omic-facts-in-lifelong-llm-agent-memory.md | 62 + ...rounding-benchmark-for-vision-language-.md | 63 + ...lling-precise-decoding-for-agentic-llms.md | 64 + ...emory-consolidation-for-language-agents.md | 61 + ...rizon-memory-environment-for-llm-agents.md | 66 + ...e-by-construction-for-generalist-agents.md | 64 + ...xploration-for-self-evolving-llm-agents.md | 63 + ...-world-small-molecule-drug-design-tasks.md | 62 + ...via-speculative-planning-for-llm-agents.md | 63 + ...-temporal-spatial-and-semantic-evasions.md | 64 + ...multi-turn-benchmark-for-agentic-safety.md | 63 + ...ts-an-empirical-study-of-curriculum-eff.md | 60 + ...uantitative-goal-persistence-in-long-ho.md | 64 + ...struments-with-natural-language-underst.md | 63 + ...ent-memory-via-causal-attribution-and-s.md | 63 + ...a-systematic-study-of-model-generated-a.md | 62 + ...tem-with-hierarchical-temporal-indexing.md | 62 + ...k-to-evaluate-mcp-poisoning-attacks-for.md | 62 + ...llm-agents-via-theory-of-mind-reasoning.md | 67 + ...level-hallucinations-in-multi-agent-ind.md | 61 + ...olar-agentic-rl-on-any-harness-at-scale.md | 61 + ...y-as-an-agent-human-interaction-problem.md | 60 + ...nst-llm-agents-via-feedback-guided-iter.md | 62 + ...nforcement-learning-for-code-generation.md | 64 + ...ing-using-edge-and-iot-data-a-review-of.md | 65 + ...-agents-on-multi-person-travel-planning.md | 61 + ...y-decodable-in-llm-agent-residual-strea.md | 60 + ...ce-aware-language-model-for-autonomous-.md | 64 + ...undamentals-attacks-and-countermeasures.md | 63 + ...onsistency-in-legal-agentic-search-thro.md | 61 + ...c-rag-under-constrained-context-budgets.md | 62 + ...a-foundations-for-long-term-ai-agent-me.md | 62 + ...5-experiments-in-agentic-ai-for-science.md | 63 + ...ion-a-dual-graph-defense-for-llm-agents.md | 60 + ...lf-evolving-llm-agents-in-cuda-kernel-g.md | 61 + ...ic-retrieval-augmented-generation-frame.md | 61 + ...ven-logical-retrieval-beyond-embeddings.md | 60 + ...ion-language-agents-for-mobile-gui-navi.md | 63 + ...-retrieval-for-emotional-support-agents.md | 62 + ...e-safety-harness-for-finance-llm-agents.md | 62 + ...kill-creation-memory-management-and-eva.md | 62 + ...i-turn-llm-agents-via-trajectory-state-.md | 63 + ...ough-contrastive-internalization-of-exp.md | 64 + ...erence-attacks-on-memory-in-chat-agents.md | 63 + ...stigation-of-layer-wise-dynamics-in-seq.md | 60 + ...y-as-cognition-in-conversational-agents.md | 64 + ...-augmented-generation-for-reliable-lega.md | 64 + ...val-augmented-generation-for-multi-agen.md | 61 + ...utilization-for-out-of-distribution-gen.md | 59 + ...mework-for-automatic-workflow-execution.md | 65 + ...-safe-agents-as-recursive-program-holes.md | 63 + ...arative-study-in-agentic-data-retrieval.md | 62 + ...or-accurate-and-generalizable-function-.md | 62 + ...eedback-alignment-in-llm-trading-agents.md | 65 + ...-memory-through-action-world-interactio.md | 63 + ...r-attributing-retrieval-lift-in-agent-m.md | 63 + ...tem-for-stateful-llm-based-applications.md | 61 + ...agents-master-pok-mon-trading-card-game.md | 58 + ...en-optimized-formats-in-agentic-ai-syst.md | 60 + ...ution-for-llm-based-multi-agent-systems.md | 61 + ...lignment-framework-for-ai-agent-safety-.md | 63 + ...ch-a-multi-agent-harness-for-interleave.md | 66 + ...tacks-through-conversational-interactio.md | 62 + ...lm-agents-exhibit-human-like-psychology.md | 63 + ...generation-with-personalized-multi-agen.md | 64 + ...ta-engineering-for-model-specialization.md | 64 + ...architecture-for-regulated-cybersecurit.md | 67 + ...-as-a-learnable-resource-for-llm-agents.md | 63 + ...icient-memory-evolution-in-agentic-llms.md | 65 + ...forecasting-with-adaptive-factor-memory.md | 63 + ...tive-self-evolving-agentic-jailbreaking.md | 65 + ...ng-llm-agents-on-financial-spreadsheets.md | 62 + ...es-a-multi-agent-framework-for-evidence.md | 62 + ...used-memorization-for-multimodal-agents.md | 61 + ...026-2605-31268-mellum2-technical-report.md | 62 + ...nce-the-glide-library-for-reliable-gena.md | 61 + ...-diagnosing-and-improving-agent-traject.md | 62 + ...-tree-for-time-sensitive-news-retrieval.md | 63 + ...00198-bagen-are-llm-agents-budget-aware.md | 63 + ...vior-arising-from-ordinary-computer-use.md | 63 + ...em-for-graph-retrieval-augmented-genera.md | 65 + ...mpression-for-long-horizon-agent-safety.md | 60 + ...ic-memory-systems-as-evolvable-programs.md | 63 + ...or-forward-looking-ai-research-judgment.md | 59 + ...irculation-for-long-horizon-llm-agents-.md | 64 + ...-agent-decisions-against-their-defaults.md | 63 + ...autonomous-agentic-design-for-photonics.md | 61 + ...-agent-for-generalizable-radiotherapy-t.md | 63 + ...mo-with-disagree-or-commit-deliberation.md | 63 + ...ts-learn-from-experience-via-latent-rag.md | 63 + ...wire-format-for-agent-memory-operations.md | 63 + ...ld-threats-to-safer-computer-use-agents.md | 63 + ...entric-optimization-of-lakehouse-agents.md | 63 + ...in-long-horizon-organizational-dynamics.md | 64 + ...-task-decomposition-for-beyond-5g-auto-.md | 62 + ...lti-agent-orchestration-with-external-k.md | 65 + ...liable-tool-augmented-large-language-mo.md | 65 + ...xploration-learning-via-novelty-signals.md | 62 + ...-rag-for-technical-literature-reasoning.md | 67 + ...mplex-task-dependencies-and-human-align.md | 60 + ...cal-autoresearch-with-agentic-ai-models.md | 61 + ...-evaluation-for-generative-enterprise-r.md | 63 + ...thesis-for-evaluating-autonomous-agents.md | 64 + ...odels-and-agent-policies-for-llm-agents.md | 64 + ...gic-deception-in-agents-via-plan-action.md | 63 + ...odeling-co-training-for-language-agents.md | 61 + ...t-benchmark-grounded-in-korean-contexts.md | 59 + ...f-continual-learning-in-language-agents.md | 66 + ...time-series-forecasting-with-llm-agents.md | 62 + ...ystem-for-patient-trajectory-modeling-i.md | 64 + ...r-evaluating-abstention-competence-in-a.md | 62 + ...raining-harnesses-for-autonomous-agenti.md | 61 + ...ion-in-llm-agents-with-information-gain.md | 61 + ...linical-decision-making-with-large-lang.md | 60 + ...self-supervised-context-memory-training.md | 63 + ...ts-with-answer-conditioned-information-.md | 61 + ...poral-memory-system-for-embodied-agents.md | 63 + ...ocialized-evolution-in-agent-ecosystems.md | 62 + ...-an-agentic-benchmark-for-novel-api-acq.md | 61 + ...ility-controlled-self-evolving-llm-agen.md | 64 + ...reinforcement-learning-for-agent-safety.md | 62 + ...nitive-memory-for-conversational-agents.md | 63 + ...of-intervention-timing-why-affect-based.md | 63 + ...entic-memory-systems-diagnostics-and-a-.md | 62 + ...y-segment-trees-for-long-horizon-agents.md | 62 + ...-inspired-agentic-system-for-industrial.md | 63 + ...h-priority-aware-runtime-transformation.md | 59 + ...-for-person-understanding-in-llm-agents.md | 63 + ...mework-for-planning-capabilities-in-llm.md | 61 + ...idence-tracing-and-execution-provenance.md | 65 + ...h-agents-measuring-performance-inflatio.md | 61 + ...for-verifiable-reinforcement-learning-o.md | 63 + ...-early-failure-alerting-in-dialogs-and-.md | 65 + ...l-intelligence-for-clinical-literature-.md | 63 + ...nchmark-for-evaluating-llms-in-patient-.md | 62 + ...development-kits-via-llm-as-a-developer.md | 61 + ...for-off-policy-evaluation-of-llm-agents.md | 62 + ...-in-large-language-model-agents-under-w.md | 61 + ...tive-study-on-structured-and-multi-hop-.md | 61 + ...ime-adaptive-memory-for-language-agents.md | 63 + ...ent-communication-in-llm-based-multi-ag.md | 63 + ...emediation-a-guardrail-feedback-driven-.md | 63 + ...hy-memory-search-for-personal-ai-agents.md | 60 + ...ecution-state-management-for-long-horiz.md | 62 + ...set-of-action-level-mental-model-annota.md | 64 + ...-investigating-collaborative-competence.md | 64 + ...implications-of-stateful-long-horizon-w.md | 63 + ...hmark-everything-everywhere-all-at-once.md | 61 + ...tomated-machine-learning-algorithm-disc.md | 64 + ...for-llm-based-quantum-software-debuggin.md | 64 + ...-preventing-cheating-via-capped-evaluat.md | 63 + ...y-for-realistic-user-agent-interactions.md | 62 + ...d-to-end-autonomous-scientific-research.md | 62 + ...ary-propagation-failures-in-vision-lang.md | 64 + ...mplete-ultra-long-horizon-software-work.md | 65 + ...ry-adaptive-memory-for-cross-llm-agents.md | 62 + ...-quasiparticle-and-excitonic-properties.md | 66 + ...the-cold-start-safety-gap-in-llm-agents.md | 60 + ...ntropy-principle-and-the-inevitable-dis.md | 59 + ...afety-gating-civility-steering-and-affe.md | 65 + ...-integrating-cognition-culture-values-a.md | 64 + ...i-agent-coordination-in-language-agents.md | 66 + ...on-and-safety-evaluation-framework-for-.md | 62 + ...iteria-rubrics-across-the-evolving-llm-.md | 62 + ...on-native-clearing-for-agentic-commerce.md | 62 + ...rks-with-adversarial-hacker-fixer-loops.md | 62 + ...imization-via-an-fea-ai-hybrid-approach.md | 64 + ...ibution-for-silent-failures-in-llm-agen.md | 61 + ...with-memory-augmented-social-simulation.md | 63 + ...owledge-into-reusable-skills-for-agents.md | 64 + ...ous-web-navigation-grounded-in-human-br.md | 64 + ...hmark-for-computer-use-agents-with-hybr.md | 61 + ...-real-world-cloud-environments-via-dist.md | 61 + ...tive-memory-system-for-self-evolving-ll.md | 63 + ...claw-clawing-back-control-of-llm-agents.md | 62 + ...age-for-structured-3d-indoor-scene-gene.md | 62 + ...for-personally-intelligent-phone-agents.md | 62 + ...-with-lightweight-coding-agent-adapters.md | 65 + ...characterizing-false-success-in-llm-age.md | 60 + ...ised-policy-optimization-for-llm-agents.md | 61 + ...ext-engineering-for-long-horizon-tool-u.md | 64 + ...lipping-encoding-subspace-in-llm-agents.md | 63 + ...agent-for-spreadsheet-manipulation-and-.md | 63 + ...vidence-grounded-muon-collider-analysis.md | 66 + ...nt-benchmarking-for-realistic-scenarios.md | 59 + ...able-and-efficient-generalist-web-agent.md | 63 + ...on-folding-for-long-horizon-llm-agent-l.md | 62 + ...e-memory-for-long-horizon-llm-reasoning.md | 62 + ...ge-navigation-as-a-tool-calling-harness.md | 62 + ...afe-memory-retention-via-constrained-op.md | 63 + ...ocuments-for-long-term-llm-agent-memory.md | 62 + ...i-agent-llm-training-with-cross-agent-l.md | 62 + ...ng-of-multimodal-memories-in-web-agents.md | 64 + ...urfaces-attacks-defenses-and-evaluation.md | 65 + ...on-demand-hypergraph-memory-for-long-do.md | 64 + ...g-to-adapt-to-unfamiliar-programming-la.md | 61 + ...ion-of-computer-use-agentic-tasks-in-re.md | 62 + ...grounded-critic-for-computer-use-agents.md | 64 + ...simulation-toolkit-for-agent-evaluation.md | 61 + ...-framework-for-efficient-agentic-reinfo.md | 64 + ...data-into-verifiable-multimodal-stories.md | 66 + ...cation-for-hierarchical-language-agents.md | 62 + ...an-building-interaction-via-programmati.md | 64 + ...-memory-navigation-for-efficient-agents.md | 66 + ...ion-firewall-for-unattended-long-horizo.md | 62 + ...ta-a-benchmark-for-clinical-tool-agents.md | 63 + ...-building-custom-ai-agents-from-substra.md | 62 + ...ls-with-multimodal-contextual-reasoning.md | 63 + ...untime-governance-of-production-ai-agen.md | 67 + ...dgets-for-privacy-preserving-llm-agents.md | 61 + ...-openclaw-style-agent-harnesses-on-codi.md | 61 + ...-agentic-procedural-policy-optimization.md | 61 + ...a-cognition-layer-for-autonomous-agents.md | 61 + ...ger-leakage-in-vision-language-agentic-.md | 62 + ...ided-credit-distillation-for-long-horiz.md | 64 + ...or-human-mobility-trajectory-generation.md | 65 + ...table-tool-workflows-for-compact-agents.md | 62 + ...mory-poisoning-in-persistent-llm-agent-.md | 63 + ...rld-models-for-self-evolving-llm-agents.md | 64 + ...ch-agents-beyond-the-human-difficulty-c.md | 61 + ...rounded-multi-factor-value-model-for-ag.md | 64 + ...on-over-heterogeneous-earth-system-data.md | 64 + ...-compression-for-long-term-agent-memory.md | 64 + ...-multimodal-llms-task-benchmark-and-app.md | 62 + ...ogy-aware-skill-self-evolution-for-llm-.md | 60 + ...ompt-injection-benchmarking-for-real-wo.md | 65 + ...on-of-ai-agents-on-epigenomics-analysis.md | 59 + ...or-openness-standardization-and-reprodu.md | 63 + ...26-2606-13643-recursive-agent-harnesses.md | 63 + ...se-tool-calls-for-tool-augmented-agents.md | 61 + ...y-under-e-commerce-deceptive-interfaces.md | 61 + ...s-for-qa-agents-over-massive-data-lakes.md | 64 + ...safety-against-decomposition-attacks-wi.md | 64 + ...ough-a-failure-mode-study-of-gui-agents.md | 63 + ...g-and-agent-memory-you-can-replay-diff-.md | 64 + ...adigm-shift-toward-persistent-autonomou.md | 62 + ...e-attacks-on-llm-based-agent-guardrails.md | 63 + ...ent-memory-for-future-oriented-assistan.md | 60 + ...m-executable-planning-with-a-world-mode.md | 64 + ...system-for-reliable-multi-agent-workflo.md | 65 + ...lay-debugging-of-multi-agent-llm-traces.md | 61 + ...s-worth-their-tokens-a-budget-constrain.md | 62 + ...hmark-for-safety-in-computer-use-agents.md | 61 + ...ent-and-instant-agentic-intelligence-at.md | 64 + ...ual-social-intelligence-in-multimodal-s.md | 61 + ...n-security-risks-in-agent-skill-ecosyst.md | 62 + ...urrency-control-for-multi-agent-systems.md | 63 + ...ed-equation-chains-a-controlled-generat.md | 62 + ...e-language-model-agents-via-memory-base.md | 62 + ...complementary-collaboration-in-minecraf.md | 62 + ...twork-management-with-proof-of-concept-.md | 64 + ...soning-and-coherent-decision-making-of-.md | 62 + ...e-agentic-programming-for-agent-harness.md | 60 + ...-an-architectural-study-of-agent-memory.md | 62 + ...dence-for-agentic-multimodal-rag-in-lon.md | 64 + ...tem-for-therapeutic-reasoning-over-hist.md | 63 + ...kload-migration-via-in-context-learning.md | 61 + ...ents-with-pareto-ranking-policy-optimiz.md | 62 + ...rsonalized-agent-for-the-physical-world.md | 63 + ...playbooks-for-agentic-security-auditing.md | 64 + ...ontextual-grounding-for-language-agents.md | 62 + ...py-controllable-narrative-script-genera.md | 60 + ...entic-ai-generated-julia-code-on-superc.md | 61 + ...evidence-from-agentic-automata-learning.md | 62 + ...ble-active-tool-discovery-in-llm-agents.md | 62 + ...-agents-in-heterogeneous-multi-agent-ec.md | 62 + ...-language-models-for-sms-to-webpage-fra.md | 61 + ...sonally-intelligent-computer-use-agents.md | 62 + ...earch-for-agentic-large-language-models.md | 63 + ...gents-for-scientific-instrument-control.md | 62 + ...al-minimal-tool-filtering-in-llm-agents.md | 60 + ...-software-tool-discovery-case-for-log-a.md | 62 + ...ridge-scalable-evaluation-for-ai-agents.md | 62 + ...analysis-articles-from-nature-portfolio.md | 63 + ...es-computes-and-self-reviews-climate-sc.md | 64 + ...ol-using-llm-agents-in-realistic-scenar.md | 63 + ...nts-for-operational-disaster-geo-intell.md | 65 + ...s-architecture-key-mechanisms-and-proto.md | 63 + ...pomdp-based-framework-for-belief-state-.md | 63 + ...nergy-based-retrieval-augmented-generat.md | 67 + ...-aware-map-agents-through-behavior-grou.md | 59 + ...esource-reallocation-with-multi-role-ag.md | 65 + ...-transactions-for-tool-using-llm-agents.md | 63 + ...amics-in-agentic-reinforcement-learning.md | 62 + ...efficient-test-time-computation-scaling.md | 63 + ...y-verification-for-mcp-based-llm-agents.md | 64 + ...m-agents-decompose-retrieve-and-compose.md | 61 + ...-premature-diagnostic-handoff-and-silen.md | 61 + ...lfight-an-agentic-benchmark-for-implici.md | 61 + ...ents-for-energy-efficient-6g-autonomous.md | 63 + ...vidence-and-sandbox-harm-in-tool-using-.md | 63 + ...ersal-harness-for-embodied-manipulation.md | 64 + ...uided-distillation-for-long-term-memory.md | 63 + ...agentic-ai-under-retrieval-and-tool-use.md | 62 + ...ment-of-multi-agent-systems-for-enterpr.md | 62 + ...y-detection-via-specification-inference.md | 63 + ...ent-trajectories-for-interactive-verifi.md | 61 + ...c-ai-in-power-system-steady-state-studi.md | 64 + ...in-multi-principal-shared-memory-agents.md | 63 + ...gic-reasoning-by-vision-language-models.md | 64 + ...-via-suspicious-api-knowledge-and-agent.md | 62 + ...e-compliance-verification-for-ai-agents.md | 59 + ...-on-small-molecule-preclinical-pharmaco.md | 64 + ...entered-runtime-state-for-agent-systems.md | 64 + ...untime-governance-of-agentic-ai-systems.md | 63 + ...oding-agents-over-100-interaction-turns.md | 61 + ...lidity-for-the-evaluation-of-llm-agents.md | 61 + ...ging-operations-research-tasks-end-to-e.md | 60 + ...tration-for-ai-assisted-legal-discovery.md | 65 + ...s-workflows-for-lung-pathology-extracti.md | 62 + ...cal-capabilities-and-risks-of-ai-agents.md | 64 + ...obile-gui-agent-with-proactive-context-.md | 61 + ...r-mobile-gui-agents-with-hierarchical-f.md | 60 + ...licy-self-improvement-in-the-real-world.md | 62 + ...ng-over-privileged-tool-selection-in-ll.md | 62 + ...or-model-grounded-economic-analysis-wit.md | 63 + ...on-as-a-pluggable-engine-for-llm-agents.md | 61 + ...b-issue-resolution-via-multi-agent-llms.md | 64 + ...ntic-ai-in-power-system-dynamic-studies.md | 66 + ...model-guided-automated-attacks-on-agent.md | 62 + ...lures-in-vision-language-agents-via-tra.md | 62 + ...robabilistic-verification-for-ai-agents.md | 62 + ...f-repository-guidance-for-coding-agents.md | 64 + ...cits-reasoning-for-spatial-intelligence.md | 62 + ...kflow-design-for-global-agentic-collabo.md | 61 + ...the-cost-accuracy-frontier-of-llm-agent.md | 65 + ...or-vulnerability-detection-in-web-agent.md | 65 + ...ng-environments-for-computer-use-agents.md | 63 + ...-agents-against-tool-description-poison.md | 61 + ...evaluation-of-ai-agents-in-electric-pow.md | 61 + ...agent-memory-from-a-few-kilobytes-of-le.md | 62 + ...astructure-for-future-event-forecasting.md | 63 + ...akes-reasoning-evaluation-and-interpret.md | 62 + ...ting-system-architecture-for-autonomous.md | 61 + ...2606-21228-sakana-fugu-technical-report.md | 61 + ...eduling-for-low-latency-agentic-systems.md | 62 + ...e-feedback-breaks-tool-using-llm-agents.md | 61 + ...-systems-with-primitive-representations.md | 62 + ...on-for-multi-hop-qa-with-a-local-7b-mod.md | 62 + ...a-building-blocks-towards-design-time-v.md | 62 + ...ta-evaluation-dataset-for-agentic-tasks.md | 62 + ...r-long-context-retrieval-and-agentic-me.md | 62 + ...extual-privacy-alignment-for-llm-agents.md | 63 + ...-the-compression-boundary-of-llm-agents.md | 62 + ...proach-to-end-to-end-pddl-planning-with.md | 62 + ...-architectural-design-space-exploration.md | 62 + ...l-attacks-on-non-prefix-kv-cache-in-rag.md | 62 + ...ill-of-materials-for-agentic-ai-systems.md | 66 + ...ixed-language-mobile-crashes-at-industr.md | 63 + ...-world-model-for-long-term-agent-memory.md | 64 + ...work-for-repository-level-code-generati.md | 63 + ...-of-agentic-program-repair-trajectories.md | 64 + ...-research-contributions-through-structu.md | 62 + ...ety-vulnerability-detection-for-reposit.md | 64 + ...riven-skill-optimization-for-llm-agents.md | 64 + ...ning-of-llm-tool-use-agents-in-large-sc.md | 62 + ...al-codebase-index-inside-a-coding-agent.md | 61 + ...d-human-oversight-for-agentic-code-gene.md | 62 + ...tic-ai-needs-deterministic-environments.md | 62 + ...g-ai-agents-on-real-world-macos-desktop.md | 65 + ...s-research-and-human-in-the-loop-refine.md | 61 + ...-rag-for-automated-vulnerability-repair.md | 64 + ...ia-mechanistic-subspaces-for-multi-turn.md | 62 + ...ss-discipline-in-autonomous-ai-coding-a.md | 63 + ...r-llm-as-judge-in-stateful-agent-evalua.md | 62 + ...n-of-llm-agent-dependency-and-execution.md | 60 + ...nstatement-for-long-term-agentic-memory.md | 62 + ...evaluation-protocol-for-hidden-state-pr.md | 60 + ...fied-search-for-long-horizon-gui-agents.md | 63 + ...nagement-is-load-bearing-for-llm-agents.md | 61 + ...ial-analysts-beyond-finance-agent-v2-wi.md | 62 + ...-in-security-of-vibe-coded-applications.md | 64 + ...tion-of-evaluator-bias-via-agent-memory.md | 63 + ...hancing-implicit-logical-memory-retriev.md | 60 + ...ork-for-video-understanding-and-editing.md | 63 + ...al-prompts-in-populations-of-llm-agents.md | 61 + ...-agent-framework-with-3d-spatial-memory.md | 66 + ...ization-improve-multi-agent-llm-systems.md | 61 + ...ry-layer-for-continuity-handoff-and-cur.md | 60 + ...cieties-from-collective-affect-to-autho.md | 62 + ...amic-red-teaming-for-agentic-ai-systems.md | 61 + ...2026-2606-23991-critique-of-agent-model.md | 67 + ...d-multi-agent-drl-framework-for-low-alt.md | 63 + ...g-agent-for-spatial-proteomics-analysis.md | 63 + ...st-poisoning-non-malleable-origin-bound.md | 60 + ...-poisoning-effects-on-ai-security-agent.md | 62 + ...tion-of-policy-driven-physical-layer-sy.md | 61 + ...mory-sustains-mixture-of-agents-scaling.md | 61 + ...4453-bayesian-control-for-coding-agents.md | 62 + ...r-use-agents-with-autonomous-evaluation.md | 61 + ...arison-as-process-reward-for-gui-agents.md | 62 + ...ared-memory-for-multi-agent-llm-systems.md | 62 + ...n-only-and-skill-mediated-computer-use-.md | 64 + ...t-memory-via-hidden-user-state-recovery.md | 61 + ...anguage-world-models-for-general-agents.md | 63 + ...mantic-rewriting-achieving-confidential.md | 63 + ...lt-attribution-via-active-investigation.md | 63 + ...ognition-for-zero-shot-3d-understanding.md | 62 + ...ments-an-llm-based-multi-agent-approach.md | 61 + ...earning-in-supply-chain-via-contextual-.md | 61 + ...ready-for-an-agent-native-memory-system.md | 64 + ...riant-prioritization-and-diagnosis-of-g.md | 62 + ...tic-localization-for-code-repair-agents.md | 63 + ...luating-an-agentic-data-analysis-system.md | 62 + ...s-agent-data-recipes-for-agentic-models.md | 59 + ...-agentic-ai-from-foundations-to-systems.md | 68 + ...lures-in-agentic-persuasion-via-taxonom.md | 63 + ...tinual-learning-via-budget-curated-memo.md | 61 + ...-benchmarking-agentic-ai-skills-in-buil.md | 60 + ...olidation-for-llm-agents-with-long-term.md | 63 + ...-policy-enforcement-for-agent-harnesses.md | 62 + ...essment-for-cost-efficient-multi-agent-.md | 62 + ...ion-progress-pitfalls-and-paths-forward.md | 63 + ...ion-with-a-visuo-spatio-temporal-memory.md | 65 + ...le-multi-agent-framework-for-safe-and-c.md | 62 + ...lm-architecture-for-stealth-assessment-.md | 63 + ...w-different-memory-roles-shape-conversa.md | 63 + ...multi-agent-framework-for-autonomous-br.md | 62 + ...n-agentic-ai-framework-for-e-scooter-mo.md | 63 + ...ve-multi-agent-scaffolding-for-efficien.md | 64 + ...lation-as-a-hidden-cost-of-low-bit-reas.md | 63 + ...amework-for-cross-library-test-migratio.md | 64 + ...its-evaluating-multi-agent-systems-for-.md | 65 + ...-medical-error-detection-and-correction.md | 62 + ...h-agentic-solutions-with-context-optimi.md | 63 + ...d-exploration-of-user-sensitive-screens.md | 60 + ...se-agents-a-benchmark-across-vision-lan.md | 64 + ...-using-agents-under-tool-environment-un.md | 61 + ...is-multi-environment-evaluation-of-fron.md | 61 + ...me-ai-alignment-for-ai-agents-and-other.md | 64 + ...re-an-llm-powered-pipeline-for-comparat.md | 64 + ...l-health-medication-information-seeking.md | 60 + ...t-contracts-against-real-world-on-chain.md | 61 + ...-silver-bullet-for-coding-agent-rewards.md | 63 + ...rm-on-real-world-energy-analytics-tasks.md | 63 + ...ence-in-prompt-composed-agentic-systems.md | 61 + ...substrate-for-privacy-memory-and-tool-u.md | 60 + ...ing-tools-as-expert-surrogates-for-llm-.md | 63 + ...es-against-prompt-injection-in-llm-agen.md | 64 + ...minating-stale-fact-errors-for-ai-agent.md | 65 + ...ioral-specifications-in-ai-agent-skills.md | 60 + ...n-the-loop-agentic-system-for-scientifi.md | 67 + ...centric-survey-of-privacy-in-llm-agents.md | 64 + ...-agent-instructions-into-policy-as-code.md | 62 + ...orkflow-for-agent-mediated-knowledge-co.md | 67 + ...d-agent-framework-for-kernel-generation.md | 62 + ...guided-mcts-red-teaming-for-agentic-rag.md | 65 + ...parametric-consolidation-for-long-runni.md | 61 + ...socio-economic-systems-powered-by-llm-a.md | 60 + ...g-task-insensitivity-in-language-agents.md | 61 + ...tic-control-plane-for-llm-coding-agents.md | 64 + ...ng-system-administration-with-ai-agents.md | 61 + ...-stopping-for-iterative-llm-agent-loops.md | 63 + ...me-labels-to-causal-process-supervision.md | 61 + ...or-architecture-evolution-in-industrial.md | 64 + ...rience-exploration-and-hindsight-experi.md | 65 + ...led-agentic-ai-driven-hardware-software.md | 61 + ...nts-in-open-ended-positive-sum-bargaini.md | 64 + ...it-software-world-models-in-coding-llms.md | 63 + ...esearch-with-parallel-llm-coding-agents.md | 61 + ...ing-the-memory-update-gap-in-llm-agents.md | 64 + ...c-training-paradigm-for-world-model-pla.md | 62 + ...tion-topologies-for-token-efficient-llm.md | 61 + ...dal-agents-visual-memory-with-incidenta.md | 65 + ...anguage-model-for-content-and-ai-safety.md | 62 + ...d-planning-for-reliable-language-agents.md | 64 + ...towards-embodied-collective-intelligenc.md | 62 + ...ti-agent-llm-honeypot-for-ssh-deception.md | 61 + ...g-llm-agents-for-fault-tolerant-control.md | 66 + ...-bound-privacy-in-tool-using-llm-agents.md | 62 + ...-modeling-embodied-multi-agent-behavior.md | 66 + ...ions-for-optimizing-multi-agent-systems.md | 60 + ...m-architecture-taxonomy-and-engineering.md | 65 + ...sign-as-repository-level-code-evolution.md | 60 + ...emory-system-for-long-context-reasoning.md | 64 + ...ith-institutional-guardrails-for-academ.md | 63 + ...-evolving-agents-via-held-out-selection.md | 61 + ...hesizable-c-conversion-and-verification.md | 63 + ...teganography-in-multi-agent-llm-systems.md | 65 + ...r-what-you-check-not-what-you-requested.md | 60 + ...nagement-for-long-horizon-coding-agents.md | 63 + ...free-program-verifier-for-coding-agents.md | 61 + ...ve-survey-of-self-security-and-empowere.md | 61 + ...idence-from-gaslighting-ai-agents-in-a-.md | 61 + ...l-energy-anomaly-detection-and-llm-driv.md | 65 + ...for-general-purpose-terminal-use-agents.md | 63 + ...ic-framework-for-holistic-athlete-profi.md | 64 + ...ains-from-trism-guided-agentic-workflow.md | 64 + ...nfused-deputy-failures-in-llm-agent-fra.md | 60 + ...asoning-over-a-biomedical-tool-universe.md | 61 + ...agents-know-when-to-stop-instead-of-act.md | 60 + ...-28739-agent-safety-is-action-alignment.md | 60 + ...owledge-topology-for-agent-first-memory.md | 63 + ...software-engineering-and-the-evolution-.md | 65 + ...uring-output-distribution-coupling-in-m.md | 62 + ...tic-framework-with-mcp-and-proof-repair.md | 64 + ...agent-framework-for-sar-data-generation.md | 63 + ...ion-a-wildchat-benchmark-and-cost-aware.md | 65 + ...egrity-in-multi-agent-llm-collaboration.md | 61 + ...ortation-engineering-practice-a-develop.md | 63 + ...lti-agent-ai-through-runtime-monitoring.md | 62 + ...-a-study-on-multiple-choice-question-an.md | 60 + ...ntic-workflows-a-study-on-n8n-ecosystem.md | 62 + ...-practitioner-systematization-of-autono.md | 62 + ...y-retention-for-long-horizon-llm-agents.md | 62 + ...llm-agents-in-microservice-failure-diag.md | 62 + ...-verifier-for-policy-adherence-in-llm-a.md | 62 + ...ority-voting-in-multi-agent-llm-debates.md | 61 + ...315-hierarchical-experimentalist-agents.md | 64 + ...unication-for-efficient-multi-agent-rea.md | 61 + ...-tasks-via-generalized-keyframe-extract.md | 62 + ...ocess-level-social-influence-evaluation.md | 61 + ...agents-on-long-horizon-real-world-tasks.md | 67 + ...iberation-with-local-reliability-bounds.md | 64 + ...r-audit-of-evaluator-driven-preference-.md | 60 + ...framework-for-automatic-microservice-de.md | 63 + ...-language-agents-with-turn-level-credit.md | 61 + ...tform-for-medical-deep-research-with-in.md | 63 + ...ers-are-llm-agents-a-case-study-on-molt.md | 59 + ...nsistent-benchmark-for-diagnostic-evalu.md | 64 + ...emory-for-agentic-embodied-manipulation.md | 64 + ...mory-system-for-long-term-conversations.md | 63 + ...mation-leaks-in-multimodal-agent-memory.md | 59 + ...gents-with-implicit-activation-steering.md | 64 + ...ation-retrieval-evaluation-in-mathemati.md | 62 + ...en-confounds-in-agent-memory-evaluation.md | 61 + ...-long-horizon-civrealm-strategy-plannin.md | 63 + ...ing-agents-in-interactive-user-sessions.md | 61 + ...mory-agents-via-dual-space-distillation.md | 61 + ...iciting-self-managed-context-via-a-prop.md | 60 + ...-design-of-embodied-agent-architectures.md | 66 + ...m-bot-unmasking-web-agents-with-multi-l.md | 63 + ...ol-evolution-for-vision-language-agents.md | 60 + ...redit-optimization-for-agentic-tool-use.md | 61 + ...ce-llms-to-mitigate-disinformation-thre.md | 60 + ...lora-variants-for-incremental-motion-un.md | 59 + ...trations-with-real-time-voice-question-.md | 66 + ...i-party-principal-loyalty-in-llm-agents.md | 60 + ...thout-individual-fidelity-in-llm-agents.md | 61 + ...dme-md-generation-evaluating-single-age.md | 63 + ...-framework-for-reliable-multi-agent-sys.md | 62 + ...-defense-in-multi-agent-systems-routing.md | 64 + ...-coding-agent-workloads-for-llm-serving.md | 62 + ...es-for-agent-memory-poisoning-detection.md | 63 + ...s-user-driven-long-horizon-coding-sessi.md | 63 + ...n-channels-for-securing-multi-agent-sys.md | 62 + ...aching-trillion-parameter-performance-w.md | 63 + ...ing-world-models-for-llm-agent-planning.md | 65 + ...er-for-accessibility-grounded-ai-agents.md | 63 + ...ent-security-through-a-computer-systems.md | 61 + ...ction-for-iterative-prompt-optimization.md | 62 + ...rom-advanced-regulatory-control-theory-.md | 62 + ...nt-systems-for-human-aligned-mental-hea.md | 63 + ...igating-multi-agent-deliberation-in-law.md | 61 + ...w-for-hls-compatibility-and-performance.md | 64 + ...mous-ai-agents-the-agentbound-framework.md | 61 + ...-collective-intelligence-in-human-agent.md | 65 + ...ificial-life-with-autonomous-llm-agents.md | 60 + ...nchmark-and-framework-for-multi-uav-col.md | 67 + ...low-for-drug-drug-interaction-predictio.md | 63 + ...or-autoformalizing-research-mathematics.md | 62 + ...estration-and-dynamic-workflows-in-lang.md | 61 + ...e-of-realistic-agentic-healthcare-envir.md | 63 + ...l-augmented-generation-with-self-reflec.md | 62 + ...-via-structured-autoregressive-modeling.md | 66 + ...ework-for-multi-layer-agent-red-teaming.md | 59 + ...-trajectories-synthesis-for-scientific-.md | 63 + ...-for-parametric-b-rep-assembly-modeling.md | 62 + ...-dynamic-model-discovery-in-power-syste.md | 60 + ...governance-for-intelligent-industrial-m.md | 64 + ...606-31410-xiaomi-gui-0-technical-report.md | 65 + ...anguage-agents-for-incremental-3d-scene.md | 59 + ...m-passive-records-to-active-task-drivin.md | 64 + ...ontrol-using-knowledge-grounded-llm-age.md | 63 + ...of-large-language-model-vulnerabilities.md | 69 + ...nt-adaptation-of-multilingual-tool-usin.md | 63 + ...ith-selective-turn-memory-in-agentic-rl.md | 62 + ...t-agent-search-system-for-geopolitical-.md | 61 + ...to-item-fulfillment-in-agentic-shopping.md | 64 + ...sics-based-household-digital-twins-for-.md | 62 + ...xecution-time-improvement-patches-in-ja.md | 60 + ...ientific-discovery-in-plant-phenotyping.md | 60 + ...ersation-assessing-the-capacity-of-llms.md | 62 + ...gaps-in-human-and-agentic-computer-use-.md | 63 + ...rative-skill-composition-for-llm-agents.md | 62 + ...ion-signals-for-long-horizon-llm-agents.md | 63 + ...eering-the-loops-that-replace-step-by-s.md | 63 + ...ssion-for-multi-agent-code-co-synthesis.md | 64 + ...itecture-drives-language-emergence-in-l.md | 63 + ...uav-enabled-wpt-systems-in-low-altitude.md | 60 + ...-evaluator-preference-dynamics-in-llm-a.md | 62 + ...fety-and-governance-for-single-and-mult.md | 63 + ...ing-eddops-with-evaluation-drivenregist.md | 60 + ...ing-latent-design-intents-for-agentic-s.md | 60 + ...g-reasoning-in-agentic-retrieval-augmen.md | 60 + ...r-tool-augmented-scientific-simulator-a.md | 61 + ...rk-for-provenance-based-backward-tracki.md | 63 + ...-llm-for-context-aware-agricultural-adv.md | 67 + ...tion-for-long-horizon-mobile-gui-agents.md | 63 + ...alysis-for-deep-learning-framework-bugs.md | 65 + ...age-models-an-overview-and-perspectives.md | 63 + ...ark-framework-for-world-modeling-agents.md | 61 + ...ing-context-for-long-horizon-llm-agents.md | 60 + ...skills-are-written-adapted-and-maintain.md | 60 + ...multi-agent-story-generation-for-long-f.md | 62 + ...enerate-quantum-applications-for-test-o.md | 63 + ...tic-rag-pipelines-a-proof-of-concept-st.md | 62 + ...gents-with-runtime-diagnosis-from-multi.md | 62 + ...collectives-as-interpretable-substrates.md | 59 + ...r-deterministic-self-expanding-reaction.md | 62 + ...benchmarking-sycophancy-in-agent-memory.md | 62 + ...nveiling-the-fragility-of-static-traini.md | 61 + ...rning-systems-enable-self-evolving-agen.md | 61 + ...hmarks-reliably-measuring-coding-agents.md | 61 + ...ts-on-whole-repository-compatibility-re.md | 61 + ...earch-for-federated-learning-algorithms.md | 60 + ...ng-teams-an-organizational-framework-fo.md | 64 + ...-involved-agentic-permission-management.md | 61 + ...1523-multi-head-recurrent-memory-agents.md | 62 + ...ith-ontology-error-prioritized-interact.md | 63 + ...uced-representational-coupling-in-multi.md | 60 + ...s-for-static-analysis-of-agent-programs.md | 63 + ...ng-infinite-agentic-loops-in-llm-agents.md | 62 + ...gent-deliberation-under-information-asy.md | 60 + ...istant-for-hardware-security-verificati.md | 65 + ...arnesses-for-image-generation-workflows.md | 64 + ...nt-system-for-dynamic-3d-scene-creation.md | 64 + ...le-world-model-correction-for-agent-rol.md | 62 + ...tem-in-hyper-scale-microservice-systems.md | 63 + ...isk-discovery-to-evidence-grounded-veri.md | 63 + ...ork-for-automated-topology-optimization.md | 62 + ...luating-and-enhancing-agentic-skill-use.md | 64 + ...code-memory-for-repository-level-progra.md | 62 + ...l-modal-structural-reasoning-for-agenti.md | 64 + ...mory-failures-in-long-term-agent-memory.md | 60 + ...proxy-for-agentic-capability-evaluation.md | 61 + ...collaboration-for-reliable-software-dev.md | 62 + ...or-ai-agent-decisions-in-autonomous-tel.md | 63 + ...e-for-equitable-mental-wellness-support.md | 62 + ...ory-testbed-for-long-horizon-llm-agents.md | 64 + ...on-boundary-violations-in-underspecifie.md | 61 + ...gent-simplification-for-spanish-easy-to.md | 62 + ...ing-of-fdm-parts-via-multi-agent-llm-re.md | 61 + ...nal-analysis-of-open-source-multi-agent.md | 59 + ...ng-social-structure-and-latent-objectiv.md | 63 + ...lidity-audit-of-tool-calling-evaluation.md | 60 + ...-promotion-from-correlated-agent-traces.md | 63 + ...or-measuring-enforcing-and-training-pro.md | 63 + ...gents-on-multi-bug-software-maintenance.md | 60 + ...er-optimizations-evaluating-llm-agents-.md | 60 + ...r-streaming-egocentric-memory-retrieval.md | 63 + ...-ai-for-scientific-software-development.md | 65 + ...sion-making-in-agent-based-urban-mobili.md | 64 + ...-coding-agents-for-open-ended-discovery.md | 60 + ...-environment-modeling-for-agentic-tasks.md | 61 + ...composition-attack-in-llm-coding-agents.md | 63 + ...ex-medical-calculations-with-llm-agents.md | 59 + ...agentic-workflow-via-symbolic-inference.md | 61 + ...servation-compression-for-coding-agents.md | 61 + ...rch-with-multi-tool-agentic-reasoning-v.md | 62 + ...-serving-layer-for-agentic-applications.md | 61 + ...ous-agents-in-scientific-quantum-progra.md | 60 + ...-ability-of-large-language-model-agents.md | 61 + ...configurations-of-personalizable-agents.md | 64 + ...-intelligence-and-cyber-investigations-.md | 65 + ...ve-security-framework-for-wireless-pbft.md | 63 + ...elopers-feedback-to-coderabbit-reviews-.md | 61 + ...ing-to-accelerate-agentic-llm-inference.md | 64 + ...dynamic-real-time-compositional-policie.md | 62 + ...entic-test-time-training-for-llm-agents.md | 61 + ...ce-evaluation-for-enterprise-agentic-ai.md | 66 + ...g-agents-on-real-c-runtime-environments.md | 64 + ...suring-ai-agents-as-computer-architects.md | 62 + ...agent-reasoning-for-smart-city-security.md | 62 + ...scaffolding-evolution-shapes-coding-age.md | 63 + ...607-03695-social-networks-of-llm-agents.md | 59 + ...g-via-pivotal-aware-self-feedback-retry.md | 62 + ...em-self-optimizing-memory-for-ai-agents.md | 63 + ...-prompt-injection-in-personal-ai-agents.md | 61 + ...-framework-for-radiology-report-generat.md | 61 + ...-ai-agents-with-natural-language-tools-.md | 63 + ...level-jailbreak-construction-in-ide-cod.md | 62 + ...work-for-discovering-turbulence-physics.md | 61 + ...agentic-reliability-in-function-calling.md | 61 + ...-aware-memory-plane-for-lifelong-agents.md | 61 + ...scene-reasoning-with-benchmarking-and-q.md | 63 + ...lation-via-zero-shot-workflow-reasoning.md | 66 + ...ode-generation-on-repository-scale-prob.md | 60 + ...d-challenges-toward-the-internet-of-age.md | 63 + ...0-biological-motifs-for-agentic-control.md | 62 + ...-causal-thinking-of-llm-agents-in-games.md | 64 + ...ng-state-belief-reliance-on-pixels-vers.md | 61 + ...an-auditable-agentic-memory-architectur.md | 61 + ...driven-agents-for-mathematical-research.md | 63 + ...-agentic-tool-use-for-neuron-kernel-gen.md | 61 + ...ndational-model-for-physical-agentic-ai.md | 64 + ...roadmap-for-agentic-recommender-systems.md | 66 + ...ugmented-cooperative-multi-agent-reinfo.md | 63 + ...ief-divergence-in-multi-step-llm-agents.md | 61 + ...llms-for-agentic-home-energy-management.md | 61 + ...mory-substrate-for-long-lived-ai-agents.md | 62 + ...ures-a-recovery-aware-evaluation-of-dia.md | 63 + ...gnosing-tool-use-failures-in-llm-agents.md | 62 + ...ency-structure-and-merge-conflict-rates.md | 60 + ...-optimization-for-multi-turn-llm-agents.md | 62 + ...icy-optimization-for-llm-agent-training.md | 60 + ...ber-threat-intelligence-knowledge-graph.md | 64 + ...ged-reasoning-attacks-on-llm-agent-memo.md | 62 + ...rge-language-model-agents-in-healthcare.md | 62 + ...acks-are-realistic-threats-to-ai-agents.md | 63 + ...ence-and-exploitation-in-repeated-games.md | 62 + ...el-agents-in-de-idealized-real-world-en.md | 61 + ...t-programming-horizons-in-coding-agents.md | 61 + ...ent-self-evolution-via-ability-transfer.md | 63 + ...nt-of-llm-agents-via-two-timescale-meta.md | 61 + ...integrity-in-multi-user-agentic-systems.md | 62 + ...ersonal-agents-under-evolving-intent-pl.md | 63 + ...context-compaction-for-long-horizon-age.md | 60 + ...-general-purpose-verification-framework.md | 65 + ...al-augmented-generation-system-for-evid.md | 66 + ...er-agentic-ai-system-for-bioinformatics.md | 64 + ...ses-with-offline-reinforcement-learning.md | 63 + ...ntity-bound-authorization-for-ai-agents.md | 61 + ...d-emotion-in-multi-agent-software-teams.md | 61 + ...rical-taxonomy-of-mutation-patterns-in-.md | 59 + ...erizing-coding-agent-in-open-source-sof.md | 64 + ...s-extendedworking-memory-for-language-a.md | 62 + ...esearch-for-ai-coding-agents-isolation-.md | 62 + ...via-multi-stage-reasoning-with-llm-base.md | 63 + ...ion-environments-for-scalable-agentic-r.md | 61 + ...l-use-planning-and-reasoning-failures-i.md | 65 + ...avigation-learning-to-use-memory-as-a-s.md | 63 + ...simulator-for-cryogenic-fault-diagnosis.md | 61 + ...r-engine-grounded-pcb-design-automation.md | 60 + ...t-to-execution-integrity-for-llm-agents.md | 60 + ...-in-economies-of-frontier-llm-agents-a-.md | 62 + ...ng-multilingual-long-horizon-llm-agents.md | 64 + ...plying-putnam-s-social-capital-theory-w.md | 64 + ...ental-learning-back-into-ai-assisted-so.md | 68 + ...benchmark-for-efficient-web-agent-evalu.md | 65 + ...tion-evolving-for-agentic-post-training.md | 60 + ...-a-study-on-joint-decision-making-under.md | 62 + ...agent-frameworks-with-an-agentic-oracle.md | 60 + ...imization-an-adaptive-tree-structured-r.md | 61 + ...ntime-intervention-for-reliable-llm-age.md | 62 + ...nts-for-automatic-software-verification.md | 64 + ...-benchmark-with-natively-authored-russi.md | 62 + ...ting-agentic-ai-s-autonomous-model-disc.md | 61 + ...-type-aware-llm-pipelines-for-bioasq-14.md | 63 + papers/paper-insights.md | 23 +- papers/source-registry.md | 2 +- tools/collection/README.md | 39 + tools/collection/collect_arxiv.py | 425 + tools/collection/promote_arxiv_manifest.py | 101 + 980 files changed, 189464 insertions(+), 58 deletions(-) create mode 100644 collection-runs/2026-07-08-arxiv-expanded-sweep.md create mode 100644 data/arxiv-agent-candidates-2025-07-08-to-2026-07-08.json create mode 100644 data/arxiv-agent-papers-2025-07-08-to-2026-07-08.json create mode 100644 papers/corpus-summary-2026-07-08.md create mode 100644 papers/items/2025-2507-20395-mazeeval-a-benchmark-for-testing-sequential-decision-making-in-language-models.md create mode 100644 papers/items/2025-2507-20666-mimii-agent-leveraging-llms-with-function-calling-for-relative-evaluation-of-ano.md create mode 100644 papers/items/2025-2508-07575-mcptoolbench-a-large-scale-ai-agent-model-context-protocol-mcp-tool-use-benchmar.md create mode 100644 papers/items/2025-2508-11027-hell-or-high-water-evaluating-agentic-recovery-from-external-failures.md create mode 100644 papers/items/2025-2508-12685-toolace-mt-non-autoregressive-generation-for-agentic-multi-turn-interaction.md create mode 100644 papers/items/2025-2508-17094-powerchain-a-verifiable-agentic-ai-system-for-automating-distribution-grid-analy.md create mode 100644 papers/items/2025-2509-02444-appcopilot-toward-general-accurate-long-horizon-and-efficient-mobile-agent.md create mode 100644 papers/items/2025-2509-02494-gridmind-llms-powered-agents-for-power-system-analysis-and-operations.md create mode 100644 papers/items/2025-2509-08863-geojson-agents-a-multi-agent-llm-architecture-for-geospatial-analysis-function-c.md create mode 100644 papers/items/2025-2509-10769-agentarch-a-comprehensive-benchmark-to-evaluate-agent-architectures-in-enterpris.md create mode 100644 papers/items/2025-2509-13311-towards-general-agentic-intelligence-via-environment-scaling.md create mode 100644 papers/items/2025-2509-14477-ticket-bench-a-kickoff-for-multilingual-and-regionalized-agent-evaluation.md create mode 100644 papers/items/2025-2509-20998-core-full-path-evaluation-of-llm-agents-beyond-final-state.md create mode 100644 papers/items/2025-2509-26553-towards-reliable-benchmarking-a-contamination-free-controllable-evaluation-frame.md create mode 100644 papers/items/2025-2510-03847-small-language-models-for-agentic-systems-a-survey-of-architectures-capabilities.md create mode 100644 papers/items/2025-2510-04206-agentrl-scaling-agentic-reinforcement-learning-with-a-multi-turn-multi-task-fram.md create mode 100644 papers/items/2025-2510-14548-llm-agents-beyond-utility-an-open-ended-perspective.md create mode 100644 papers/items/2025-2510-18586-tokencake-a-kv-cache-centric-serving-framework-for-llm-based-multi-agent-applica.md create mode 100644 papers/items/2025-2510-21524-eu-agent-bench-measuring-illegal-behavior-of-llm-agents-under-eu-law.md create mode 100644 papers/items/2025-2510-22768-seeing-is-believing-evaluating-vision-language-model-susceptibility-in-agent-to-.md create mode 100644 papers/items/2025-2510-24645-funreason-mt-technical-report-advanced-data-synthesis-solution-for-real-world-mu.md create mode 100644 papers/items/2025-2510-26167-toolrm-towards-agentic-tool-use-reward-modeling.md create mode 100644 papers/items/2025-2511-04847-test-time-adaptation-for-llm-agents-via-environment-interaction.md create mode 100644 papers/items/2025-2511-11169-refine-and-align-confidence-calibration-through-multi-agent-interaction-in-vqa.md create mode 100644 papers/items/2025-2511-15203-taxonomy-evaluation-and-exploitation-of-ipi-centric-llm-agent-defense-frameworks.md create mode 100644 papers/items/2025-2511-22138-tinyllm-evaluation-and-optimization-of-small-language-models-for-agentic-tasks-o.md create mode 100644 papers/items/2025-2512-02605-iact-a-self-organizing-recursive-model-for-general-ai-agents-a-technical-white-p.md create mode 100644 papers/items/2025-2512-11682-medai-evaluating-txagent-s-therapeutic-agentic-reasoning-in-the-neurips-cure-ben.md create mode 100644 papers/items/2025-2512-23611-close-the-loop-synthesizing-infinite-tool-use-data-via-multi-agent-role-playing.md create mode 100644 papers/items/2025-2512-23647-nested-browser-use-learning-for-agentic-information-seeking.md create mode 100644 papers/items/2025-2512-23747-state-of-the-art-small-language-coder-model-mify-coder.md create mode 100644 papers/items/2026-2601-00268-beyond-perfect-apis-a-comprehensive-evaluation-of-llm-agents-under-real-world-ap.md create mode 100644 papers/items/2026-2601-05467-stelp-secure-transpilation-and-execution-of-llm-generated-programs.md create mode 100644 papers/items/2026-2601-06007-don-t-break-the-cache-an-evaluation-of-prompt-caching-for-long-horizon-agentic-t.md create mode 100644 papers/items/2026-2601-06606-cedar-context-engineering-for-agentic-data-science.md create mode 100644 papers/items/2026-2601-12988-paperguide-making-small-language-model-paper-reading-agents-more-efficient.md create mode 100644 papers/items/2026-2601-14652-mas-orchestra-understanding-and-improving-multi-agent-reasoning-through-holistic.md create mode 100644 papers/items/2026-2602-03117-agentdyn-are-your-agent-security-defenses-deployable-in-real-world-dynamic-envir.md create mode 100644 papers/items/2026-2602-03224-tame-a-trustworthy-test-time-evolution-of-agent-memory-with-systematic-benchmark.md create mode 100644 papers/items/2026-2602-03786-aorchestra-automating-sub-agent-creation-for-agentic-orchestration.md create mode 100644 papers/items/2026-2602-05115-socialveil-probing-social-intelligence-of-language-agents-under-communication-ba.md create mode 100644 papers/items/2026-2602-05302-piearena-ranking-and-profiling-language-agents-in-realistic-negotiation-scenario.md create mode 100644 papers/items/2026-2602-05386-spider-sense-intrinsic-risk-sensing-for-efficient-agent-defense-with-hierarchica.md create mode 100644 papers/items/2026-2602-07391-naamse-framework-for-evolutionary-security-evaluation-of-agents.md create mode 100644 papers/items/2026-2602-07652-agent-fence-mapping-security-vulnerabilities-across-deep-research-agents.md create mode 100644 papers/items/2026-2602-07962-loca-bench-benchmarking-language-agents-under-controllable-and-extreme-context-g.md create mode 100644 papers/items/2026-2602-08082-spectral-guardrails-for-agents-in-the-wild-detecting-tool-use-hallucinations-via.md create mode 100644 papers/items/2026-2602-08412-from-assistant-to-double-agent-formalizing-and-benchmarking-attacks-on-openclaw-.md create mode 100644 papers/items/2026-2602-10133-agenttrace-a-structured-logging-framework-for-agent-system-observability.md create mode 100644 papers/items/2026-2602-11749-air-improving-agent-safety-through-incident-response.md create mode 100644 papers/items/2026-2602-13379-unsafer-in-many-turns-benchmarking-and-defending-multi-turn-safety-risks-in-tool.md create mode 100644 papers/items/2026-2602-13530-remem-reasoning-with-episodic-memory-in-language-agent.md create mode 100644 papers/items/2026-2602-13665-hyfunc-accelerating-llm-based-function-calls-for-agentic-ai-through-hybrid-model.md create mode 100644 papers/items/2026-2602-14234-redsearcher-a-scalable-and-cost-efficient-framework-for-long-horizon-search-agen.md create mode 100644 papers/items/2026-2602-14281-mcpshield-a-security-cognition-layer-for-adaptive-trust-calibration-in-model-con.md create mode 100644 papers/items/2026-2602-16931-narrow-fine-tuning-erodes-safety-alignment-in-vision-language-agents.md create mode 100644 papers/items/2026-2602-18456-beyond-single-channel-agentic-benchmarking.md create mode 100644 papers/items/2026-2602-19008-capable-but-unreliable-canonical-path-deviation-as-a-causal-mechanism-of-agent-f.md create mode 100644 papers/items/2026-2602-21127-are-you-sure-an-empirical-study-of-human-perception-vulnerability-in-llm-driven-.md create mode 100644 papers/items/2026-2602-23320-parammem-augmenting-language-agents-with-parametric-reflective-memory.md create mode 100644 papers/items/2026-2603-00131-thought-virus-viral-misalignment-via-subliminal-prompting-in-multi-agent-systems.md create mode 100644 papers/items/2026-2603-00623-tracesir-a-multi-agent-framework-for-structured-analysis-and-reporting-of-agenti.md create mode 100644 papers/items/2026-2603-00801-the-synthetic-web-adversarially-curated-mini-internets-for-diagnosing-epistemic-.md create mode 100644 papers/items/2026-2603-01438-enhancing-persona-following-at-decoding-time-via-dynamic-importance-estimation-f.md create mode 100644 papers/items/2026-2603-01712-ft-dojo-towards-autonomous-llm-fine-tuning-with-language-agents.md create mode 100644 papers/items/2026-2603-02711-a-natural-language-agentic-approach-to-study-affective-polarization.md create mode 100644 papers/items/2026-2603-03515-the-controllability-trap-a-governance-framework-for-military-ai-agents.md create mode 100644 papers/items/2026-2603-03680-mage-meta-reinforcement-learning-for-language-agents-toward-strategic-exploratio.md create mode 100644 papers/items/2026-2603-05553-eigendata-a-self-evolving-multi-agent-platform-for-function-calling-data-synthes.md create mode 100644 papers/items/2026-2603-05578-tool-genesis-a-task-driven-tool-creation-benchmark-for-self-evolving-language-ag.md create mode 100644 papers/items/2026-2603-07496-from-thinker-to-society-security-in-hierarchical-autonomy-evolution-of-ai-agents.md create mode 100644 papers/items/2026-2603-07557-agentraft-automated-detection-of-data-over-exposure-in-llm-agents.md create mode 100644 papers/items/2026-2603-07980-onemillion-bench-how-far-are-language-agents-from-human-experts.md create mode 100644 papers/items/2026-2603-08721-kernelcraft-benchmarking-for-agentic-close-to-metal-kernel-generation-on-emergin.md create mode 100644 papers/items/2026-2603-09002-security-considerations-for-multi-agent-systems.md create mode 100644 papers/items/2026-2603-10492-human-ai-co-reasoning-for-clinical-diagnosis-with-evidence-integrated-language-a.md create mode 100644 papers/items/2026-2603-11088-the-attack-and-defense-landscape-of-agentic-ai-a-comprehensive-survey.md create mode 100644 papers/items/2026-2603-11890-quare-quality-aware-requirements-analysis-through-multi-agent-dialectical-negoti.md create mode 100644 papers/items/2026-2603-15309-cctu-a-benchmark-for-tool-use-under-complex-constraints.md create mode 100644 papers/items/2026-2603-15666-compiled-memory-not-more-information-but-more-precise-instructions-for-language-.md create mode 100644 papers/items/2026-2603-16734-differential-harm-propensity-in-personalized-llm-agents-the-curious-case-of-ment.md create mode 100644 papers/items/2026-2603-17392-agentic-cognitive-profiling-realigning-automated-alzheimer-s-disease-detection-w.md create mode 100644 papers/items/2026-2603-18245-who-tests-the-testers-systematic-enumeration-and-coverage-audit-of-llm-agent-too.md create mode 100644 papers/items/2026-2603-19469-a-framework-for-formalizing-llm-agent-security.md create mode 100644 papers/items/2026-2603-19684-tsegagent-zero-shot-tooth-segmentation-via-geometry-aware-vision-language-agents.md create mode 100644 papers/items/2026-2603-21357-agenther-hindsight-experience-replay-for-llm-agent-trajectory-relabeling.md create mode 100644 papers/items/2026-2603-21564-toward-a-theory-of-hierarchical-memory-for-language-agents.md create mode 100644 papers/items/2026-2603-24257-memory-augmented-vision-language-agents-for-persistent-and-semantically-consiste.md create mode 100644 papers/items/2026-2603-25353-safeguard-asf-sr-agentic-humanoid-robot-system-for-autonomous-industrial-safety.md create mode 100644 papers/items/2026-2603-27148-safetydrift-predicting-when-ai-agents-cross-the-line-before-they-actually-do.md create mode 100644 papers/items/2026-2603-28166-evaluating-privilege-usage-of-agents-with-real-world-tools.md create mode 100644 papers/items/2026-2603-28428-synergy-a-next-generation-general-purpose-agent-for-open-agentic-web.md create mode 100644 papers/items/2026-2603-28900-robust-multi-agent-reinforcement-learning-for-small-uas-separation-assurance-und.md create mode 100644 papers/items/2026-2604-02022-atbench-a-diverse-and-realistic-agent-trajectory-benchmark-for-safety-evaluation.md create mode 100644 papers/items/2026-2604-02155-brief-is-better-non-monotonic-chain-of-thought-budget-effects-in-function-callin.md create mode 100644 papers/items/2026-2604-03098-co-evolution-of-policy-and-internal-reward-for-language-agents.md create mode 100644 papers/items/2026-2604-03242-draft-task-decoupled-latent-reasoning-for-agent-safety.md create mode 100644 papers/items/2026-2604-04131-profile-then-reason-bounded-semantic-complexity-for-tool-augmented-language-agen.md create mode 100644 papers/items/2026-2604-04426-shieldnet-network-level-guardrails-against-emerging-supply-chain-injections-in-a.md create mode 100644 papers/items/2026-2604-06762-arulecon-agentic-security-rule-conversion.md create mode 100644 papers/items/2026-2604-06972-differentiable-environment-trajectory-co-optimization-for-safe-multi-agent-navig.md create mode 100644 papers/items/2026-2604-08388-awakening-the-sleeping-agent-lean-specific-agentic-data-reactivates-general-tool.md create mode 100644 papers/items/2026-2604-10577-the-blind-spot-of-agent-safety-how-benign-user-instructions-expose-critical-vuln.md create mode 100644 papers/items/2026-2604-11557-unitoolcall-unifying-tool-use-representation-data-and-evaluation-for-llm-agents.md create mode 100644 papers/items/2026-2604-12986-parallax-why-ai-agents-that-think-must-never-act.md create mode 100644 papers/items/2026-2604-13298-can-agents-secure-hardware-evaluating-agentic-llm-driven-obfuscation-for-ip-prot.md create mode 100644 papers/items/2026-2604-13536-don-t-let-ai-agents-yolo-your-files-shifting-information-and-control-to-filesyst.md create mode 100644 papers/items/2026-2604-13954-hintbench-horizon-agent-intrinsic-non-attack-trajectory-benchmark.md create mode 100644 papers/items/2026-2604-14399-spacemind-a-modular-and-self-evolving-embodied-vision-language-agent-framework-f.md create mode 100644 papers/items/2026-2604-15415-harmfulskillbench-how-do-harmful-skills-weaponize-your-agents.md create mode 100644 papers/items/2026-2604-15579-don-t-make-models-guess-security-and-safety-symbolic-guardrails-for-domain-speci.md create mode 100644 papers/items/2026-2604-16706-evaluating-tool-using-language-agents-judge-reliability-propagation-cascades-and.md create mode 100644 papers/items/2026-2604-17562-safeagent-a-runtime-protection-architecture-for-agentic-systems.md create mode 100644 papers/items/2026-2604-18658-owner-harm-a-missing-threat-model-for-ai-agent-safety.md create mode 100644 papers/items/2026-2604-18718-towards-optimal-agentic-architectures-for-offensive-security-tasks.md create mode 100644 papers/items/2026-2604-18847-human-guided-harm-recovery-for-computer-use-agents.md create mode 100644 papers/items/2026-2604-19821-jtpro-a-joint-tool-prompt-reflective-optimization-framework-for-language-agents.md create mode 100644 papers/items/2026-2604-19844-if-you-re-waiting-for-a-sign-that-might-not-be-it-mitigating-trust-boundary-conf.md create mode 100644 papers/items/2026-2604-20994-breaking-mcp-with-function-hijacking-attacks-novel-threats-for-function-calling-.md create mode 100644 papers/items/2026-2604-21190-spatio-adaptive-test-time-orchestration-of-vision-language-agents-for-spatial-re.md create mode 100644 papers/items/2026-2604-22879-beyond-single-agent-alignment-preventing-context-fragmented-violations-in-multi-.md create mode 100644 papers/items/2026-2604-23210-discovering-agentic-safety-specifications-from-1-bit-danger-signals.md create mode 100644 papers/items/2026-2604-23374-ghost-in-the-agent-redefining-information-flow-tracking-for-llm-agents.md create mode 100644 papers/items/2026-2604-23459-architecture-matters-for-multi-agent-security.md create mode 100644 papers/items/2026-2604-24212-empowering-autonomous-debugging-agents-with-efficient-dynamic-analysis.md create mode 100644 papers/items/2026-2604-24826-a-comparative-evaluation-of-ai-agent-security-guardrails.md create mode 100644 papers/items/2026-2604-25135-fama-failure-aware-meta-agentic-framework-for-open-source-llms-in-interactive-to.md create mode 100644 papers/items/2026-2604-25318-cutscene-agent-an-llm-agent-framework-for-automated-3d-cutscene-generation.md create mode 100644 papers/items/2026-2604-25555-from-crud-to-autonomous-agents-formal-validation-and-zero-trust-security-for-sem.md create mode 100644 papers/items/2026-2604-26274-enforcing-benign-trajectories-a-behavioral-firewall-for-structured-workflow-ai-a.md create mode 100644 papers/items/2026-2604-26959-careguardai-context-aware-multi-agent-guardrails-for-clinical-safety-hallucinati.md create mode 100644 papers/items/2026-2604-27092-end-to-end-autonomous-scientific-discovery-on-a-real-optical-platform.md create mode 100644 papers/items/2026-2604-27464-security-attack-and-defense-strategies-for-autonomous-agent-frameworks-a-layered.md create mode 100644 papers/items/2026-2604-27699-bridging-values-and-behavior-a-hierarchical-framework-for-proactive-embodied-age.md create mode 100644 papers/items/2026-2604-27859-rethinking-agentic-reinforcement-learning-in-large-language-models.md create mode 100644 papers/items/2026-2604-28157-flashrt-towards-computationally-and-memory-efficient-red-teaming-for-prompt-inje.md create mode 100644 papers/items/2026-2605-00081-alignment-contracts-for-agentic-security-systems.md create mode 100644 papers/items/2026-2605-00741-self-adaptive-multi-agent-llm-based-security-pattern-selection-for-iot-systems.md create mode 100644 papers/items/2026-2605-00845-graph-query-generation-with-constraint-guided-large-language-agents.md create mode 100644 papers/items/2026-2605-01101-virtual-speech-therapist-a-clinician-in-the-loop-ai-speech-therapy-agent-for-per.md create mode 100644 papers/items/2026-2605-01644-toward-a-principled-framework-for-agent-safety-measurement.md create mode 100644 papers/items/2026-2605-02240-physicianbench-evaluating-llm-agents-in-real-world-ehr-environments.md create mode 100644 papers/items/2026-2605-03242-enhancing-agent-safety-judgment-controlled-benchmark-rewriting-and-analogical-re.md create mode 100644 papers/items/2026-2605-03312-memflow-intent-driven-memory-orchestration-for-small-language-model-agents.md create mode 100644 papers/items/2026-2605-03328-llm-adam-a-generalizable-llm-agent-framework-for-pre-print-anomaly-detection-in-.md create mode 100644 papers/items/2026-2605-03505-lats-rca-language-agent-tree-search-for-root-cause-analysis-in-microservices.md create mode 100644 papers/items/2026-2605-04107-tscg-deterministic-tool-schema-compilation-for-agentic-llm-deployments.md create mode 100644 papers/items/2026-2605-04808-decodingtrust-agent-platform-dtap-a-controllable-and-interactive-red-teaming-pla.md create mode 100644 papers/items/2026-2605-05242-beyond-semantic-similarity-rethinking-retrieval-for-agentic-search-via-direct-co.md create mode 100644 papers/items/2026-2605-05704-safeharbor-hierarchical-memory-augmented-guardrail-for-llm-agent-safety.md create mode 100644 papers/items/2026-2605-05716-more-is-not-always-better-cross-component-interference-in-llm-agent-scaffolding.md create mode 100644 papers/items/2026-2605-06078-milestone-guided-policy-learning-for-long-horizon-language-agents.md create mode 100644 papers/items/2026-2605-06713-agentic-ai-and-the-industrialization-of-cyber-offense-forecast-consequences-and-.md create mode 100644 papers/items/2026-2605-06716-from-storage-to-experience-a-survey-on-the-evolution-of-llm-agent-memory-mechani.md create mode 100644 papers/items/2026-2605-06737-a-self-healing-framework-for-reliable-llm-based-autonomous-agents.md create mode 100644 papers/items/2026-2605-06812-towards-security-auditable-llm-agents-a-unified-graph-representation.md create mode 100644 papers/items/2026-2605-06869-agentick-a-unified-benchmark-for-general-sequential-decision-making-agents.md create mode 100644 papers/items/2026-2605-06890-beyond-the-black-box-interpretability-of-agentic-ai-tool-use.md create mode 100644 papers/items/2026-2605-06957-learning-and-reusing-policy-decompositions-for-hierarchical-generalized-planning.md create mode 100644 papers/items/2026-2605-06992-why-does-agentic-safety-fail-to-generalize-across-tasks.md create mode 100644 papers/items/2026-2605-07112-switchcraft-ai-model-router-for-agentic-tool-calling.md create mode 100644 papers/items/2026-2605-07251-can-agents-price-a-reaction-evaluating-llms-on-chemical-cost-reasoning.md create mode 100644 papers/items/2026-2605-07830-cybiasbench-benchmarking-bias-in-llm-agents-for-cyber-attack-scenarios.md create mode 100644 papers/items/2026-2605-08374-memq-integrating-q-learning-into-self-evolving-memory-agents-over-provenance-dag.md create mode 100644 papers/items/2026-2605-08442-defense-effectiveness-across-architectural-layers-a-mechanistic-evaluation-of-pe.md create mode 100644 papers/items/2026-2605-08763-when-llms-team-up-a-coordinated-attack-framework-for-automated-cyber-intrusions.md create mode 100644 papers/items/2026-2605-08876-otora-a-unified-red-teaming-framework-for-reasoning-level-denial-of-service-in-l.md create mode 100644 papers/items/2026-2605-08964-trustworthy-ai-ensuring-reliability-and-accountability-from-models-to-agents.md create mode 100644 papers/items/2026-2605-09168-civex-causal-intervention-verification-for-language-agents.md create mode 100644 papers/items/2026-2605-09692-causal-state-binding-predicts-action-control-in-language-agents.md create mode 100644 papers/items/2026-2605-10365-agent-valuebench-a-comprehensive-benchmark-for-evaluating-agent-values.md create mode 100644 papers/items/2026-2605-10763-matra-modeling-the-attack-surface-of-agentic-ai-systems-openclaw-case-study.md create mode 100644 papers/items/2026-2605-10779-litmus-benchmarking-behavioral-jailbreaks-of-llm-agents-in-real-os-environments.md create mode 100644 papers/items/2026-2605-10870-remember-the-decision-not-the-description-a-rate-distortion-framework-for-agent-.md create mode 100644 papers/items/2026-2605-11039-the-granularity-mismatch-in-agent-security-argument-level-provenance-solves-enfo.md create mode 100644 papers/items/2026-2605-11225-pivot-bridging-planning-and-execution-in-llm-agents-via-trajectory-refinement.md create mode 100644 papers/items/2026-2605-11388-deep-reasoning-in-general-purpose-agents-via-structured-meta-cognition.md create mode 100644 papers/items/2026-2605-11534-prism-planning-and-reasoning-with-intent-in-simulated-embodied-environments.md create mode 100644 papers/items/2026-2605-11633-can-llm-agents-respond-to-disasters-benchmarking-heterogeneous-geospatial-reason.md create mode 100644 papers/items/2026-2605-11882-on-policy-self-evolution-via-failure-trajectories-for-agentic-safety-alignment.md create mode 100644 papers/items/2026-2605-11928-when-simulation-lies-a-sim-to-real-benchmark-and-domain-randomized-rl-recipe-for.md create mode 100644 papers/items/2026-2605-11946-counterfactual-trace-auditing-of-llm-agent-skills.md create mode 100644 papers/items/2026-2605-12015-skillsafetybench-evaluating-agent-safety-under-skill-facing-attack-surfaces.md create mode 100644 papers/items/2026-2605-12061-sage-a-self-evolving-agentic-graph-memory-engine-for-structure-aware-associative.md create mode 100644 papers/items/2026-2605-12260-prism-pareto-efficient-retrieval-over-intent-aware-structured-memory-for-long-ho.md create mode 100644 papers/items/2026-2605-13481-personalai-2-0-enhancing-knowledge-graph-traversal-retrieval-with-planning-mecha.md create mode 100644 papers/items/2026-2605-13542-realicu-do-llm-agents-understand-long-context-icu-data-a-benchmark-beyond-behavi.md create mode 100644 papers/items/2026-2605-13618-openaaas-an-open-agent-as-a-service-framework-for-distributed-materials-informat.md create mode 100644 papers/items/2026-2605-13716-skillops-managing-llm-agent-skill-libraries-as-self-maintaining-software-ecosyst.md create mode 100644 papers/items/2026-2605-14126-reinforcement-learning-for-tool-calling-agents-in-fast-healthcare-interoperabili.md create mode 100644 papers/items/2026-2605-14290-web-agents-should-adopt-the-plan-then-execute-paradigm.md create mode 100644 papers/items/2026-2605-14322-are-agents-ready-to-teach-a-multi-stage-benchmark-for-real-world-teaching-workfl.md create mode 100644 papers/items/2026-2605-14421-memlineage-lineage-guided-enforcement-for-llm-agent-memory.md create mode 100644 papers/items/2026-2605-14460-exploiting-llm-agent-supply-chains-via-payload-less-skills.md create mode 100644 papers/items/2026-2605-14498-groupmembench-benchmarking-llm-agent-memory-in-multi-party-conversations.md create mode 100644 papers/items/2026-2605-14527-lang2mlip-end-to-end-language-to-machine-learning-interatomic-potential-developm.md create mode 100644 papers/items/2026-2605-14892-beyond-individual-intelligence-surveying-collaboration-failure-attribution-and-s.md create mode 100644 papers/items/2026-2605-14906-memlens-benchmarking-multimodal-long-term-memory-in-large-vision-language-models.md create mode 100644 papers/items/2026-2605-14932-toward-securing-ai-agents-like-operating-systems.md create mode 100644 papers/items/2026-2605-15040-orchard-an-open-source-agentic-modeling-framework.md create mode 100644 papers/items/2026-2605-15128-memeye-a-visual-centric-evaluation-framework-for-multimodal-agent-memory.md create mode 100644 papers/items/2026-2605-15206-agentstop-terminating-local-ai-agents-early-to-save-energy-in-consumer-devices.md create mode 100644 papers/items/2026-2605-15625-colpackagent-agent-skill-guided-hard-particle-monte-carlo-workflows-for-colloida.md create mode 100644 papers/items/2026-2605-15701-h-mem-a-novel-memory-mechanism-for-evolving-and-retrieving-agent-memory-via-a-hy.md create mode 100644 papers/items/2026-2605-15710-smmbench-a-benchmark-for-source-distributed-multimodal-agent-memory.md create mode 100644 papers/items/2026-2605-15759-dimmem-dimensional-structuring-for-efficient-long-term-agent-memory.md create mode 100644 papers/items/2026-2605-16233-forge-self-evolving-agent-memory-with-no-weight-updates-via-population-broadcast.md create mode 100644 papers/items/2026-2605-16481-visual-agentic-memory-enabling-online-long-video-understanding-via-online-indexi.md create mode 100644 papers/items/2026-2605-16821-multi-paradigm-agent-interaction-in-practice-a-systematic-analysis-of-generator-.md create mode 100644 papers/items/2026-2605-17075-a-red-teaming-framework-for-evaluating-robustness-of-ai-enabled-security-orchest.md create mode 100644 papers/items/2026-2605-17348-taming-zombie-agents-a-markov-state-aware-framework-for-resilient-multi-agent-ev.md create mode 100644 papers/items/2026-2605-17453-trust-no-tool-evaluating-and-defending-llm-agents-under-untrusted-tool-feedback.md create mode 100644 papers/items/2026-2605-17625-episodic-semantic-memory-architecture-for-long-horizon-scientific-agents.md create mode 100644 papers/items/2026-2605-18284-commitdistill-a-lightweight-knowledge-centric-memory-layer-for-software-reposito.md create mode 100644 papers/items/2026-2605-18502-the-distance-based-formation-controller-design-for-multi-agent-systems-in-port-h.md create mode 100644 papers/items/2026-2605-18652-mementogui-learning-agentic-multimodal-memory-control-for-long-horizon-gui-agent.md create mode 100644 papers/items/2026-2605-18672-position-a-three-layer-probabilistic-assume-guarantee-architecture-is-structural.md create mode 100644 papers/items/2026-2605-18930-oep-poisoning-self-evolving-llm-agents-via-locally-correct-but-non-transferable-.md create mode 100644 papers/items/2026-2605-19604-formal-skill-programmable-runtime-skills-for-efficient-and-accurate-llm-agents.md create mode 100644 papers/items/2026-2605-19952-rethinking-how-to-remember-beyond-atomic-facts-in-lifelong-llm-agent-memory.md create mode 100644 papers/items/2026-2605-20306-wildroadbench-a-wild-aerial-road-damage-grounding-benchmark-for-vision-language-.md create mode 100644 papers/items/2026-2605-20315-mix-quant-quantized-prefilling-precise-decoding-for-agentic-llms.md create mode 100644 papers/items/2026-2605-20616-auto-dreamer-learning-offline-memory-consolidation-for-language-agents.md create mode 100644 papers/items/2026-2605-20833-memgym-a-long-horizon-memory-environment-for-llm-agents.md create mode 100644 papers/items/2026-2605-20874-governance-by-construction-for-generalist-agents.md create mode 100644 papers/items/2026-2605-21240-apex-autonomous-policy-exploration-for-self-evolving-llm-agents.md create mode 100644 papers/items/2026-2605-21740-smdd-bench-can-llms-solve-real-world-small-molecule-drug-design-tasks.md create mode 100644 papers/items/2026-2605-22154-idlespec-exploiting-idle-time-via-speculative-planning-for-llm-agents.md create mode 100644 papers/items/2026-2605-22321-benchmarking-autonomous-agents-against-temporal-spatial-and-semantic-evasions.md create mode 100644 papers/items/2026-2605-22643-boiling-the-frog-a-multi-turn-benchmark-for-agentic-safety.md create mode 100644 papers/items/2026-2605-23067-what-training-data-teaches-rl-memory-agents-an-empirical-study-of-curriculum-eff.md create mode 100644 papers/items/2026-2605-23574-push-your-agent-measuring-and-enforcing-quantitative-goal-persistence-in-long-ho.md create mode 100644 papers/items/2026-2605-23636-rf-instrument-agent-rfia-empowering-rf-instruments-with-natural-language-underst.md create mode 100644 papers/items/2026-2605-23723-memaudit-post-hoc-auditing-of-poisoned-agent-memory-via-causal-attribution-and-s.md create mode 100644 papers/items/2026-2605-23899-from-raw-experience-to-skill-consumption-a-systematic-study-of-model-generated-a.md create mode 100644 papers/items/2026-2605-23986-memforest-an-efficient-agent-memory-system-with-hierarchical-temporal-indexing.md create mode 100644 papers/items/2026-2605-24069-when-the-manual-lies-a-realistic-benchmark-to-evaluate-mcp-poisoning-attacks-for.md create mode 100644 papers/items/2026-2605-24216-agent-tom-learning-to-monitor-autonomous-llm-agents-via-theory-of-mind-reasoning.md create mode 100644 papers/items/2026-2605-24219-beyond-final-answers-auditing-trajectory-level-hallucinations-in-multi-agent-ind.md create mode 100644 papers/items/2026-2605-24220-polar-agentic-rl-on-any-harness-at-scale.md create mode 100644 papers/items/2026-2605-24309-reframing-llm-agent-security-as-an-agent-human-interaction-problem.md create mode 100644 papers/items/2026-2605-24659-iterinject-indirect-prompt-injection-against-llm-agents-via-feedback-guided-iter.md create mode 100644 papers/items/2026-2605-24812-core-code-collaborative-reinforcement-learning-for-code-generation.md create mode 100644 papers/items/2026-2605-25141-llm-agent-based-renewable-energy-forecasting-using-edge-and-iot-data-a-review-of.md create mode 100644 papers/items/2026-2605-25200-grouptravelbench-benchmarking-llm-agents-on-multi-person-travel-planning.md create mode 100644 papers/items/2026-2605-25310-tool-call-dependency-structure-is-linearly-decodable-in-llm-agent-residual-strea.md create mode 100644 papers/items/2026-2605-25393-decision-making-with-lightweight-confidence-aware-language-model-for-autonomous-.md create mode 100644 papers/items/2026-2605-25435-security-of-openclaw-agents-fundamentals-attacks-and-countermeasures.md create mode 100644 papers/items/2026-2605-25920-can-llms-time-travel-enhancing-temporal-consistency-in-legal-agentic-search-thro.md create mode 100644 papers/items/2026-2605-26165-tool-schema-compression-enables-agentic-rag-under-constrained-context-budgets.md create mode 100644 papers/items/2026-2605-26252-is-agent-memory-a-database-rethinking-data-foundations-for-long-term-ai-agent-me.md create mode 100644 papers/items/2026-2605-26305-experiments-in-agentic-ai-for-science.md create mode 100644 papers/items/2026-2605-26497-aligning-provenance-with-authorization-a-dual-graph-defense-for-llm-agents.md create mode 100644 papers/items/2026-2605-26720-towards-feedback-to-plan-decisions-for-self-evolving-llm-agents-in-cuda-kernel-g.md create mode 100644 papers/items/2026-2605-26926-from-norms-to-indicators-n2i-rag-an-agentic-retrieval-augmented-generation-frame.md create mode 100644 papers/items/2026-2605-27123-rethinking-agentic-rag-toward-llm-driven-logical-retrieval-beyond-embeddings.md create mode 100644 papers/items/2026-2605-27134-scaling-benchmarking-and-reasoning-of-vision-language-agents-for-mobile-gui-navi.md create mode 100644 papers/items/2026-2605-27240-enpmr-bench-benchmarking-proactive-memory-retrieval-for-emotional-support-agents.md create mode 100644 papers/items/2026-2605-27333-finharness-an-inline-lifecycle-safety-harness-for-finance-llm-agents.md create mode 100644 papers/items/2026-2605-27366-muse-autoskill-self-evolving-agents-via-skill-creation-memory-management-and-eva.md create mode 100644 papers/items/2026-2605-27690-traces-proactive-safety-auditing-for-multi-turn-llm-agents-via-trajectory-state-.md create mode 100644 papers/items/2026-2605-27762-peam-parametric-embodied-agent-memory-through-contrastive-internalization-of-exp.md create mode 100644 papers/items/2026-2605-27825-mrmmia-membership-inference-attacks-on-memory-in-chat-agents.md create mode 100644 papers/items/2026-2605-27935-do-agents-think-deeper-a-mechanistic-investigation-of-layer-wise-dynamics-in-seq.md create mode 100644 papers/items/2026-2605-28046-memcog-from-memory-as-tool-to-memory-as-cognition-in-conversational-agents.md create mode 100644 papers/items/2026-2605-28120-legalgraphrag-multi-agent-graph-retrieval-augmented-generation-for-reliable-lega.md create mode 100644 papers/items/2026-2605-28175-mixture-of-experts-knowledge-graph-retrieval-augmented-generation-for-multi-agen.md create mode 100644 papers/items/2026-2605-28424-skill0-5-joint-skill-internalization-and-utilization-for-out-of-distribution-gen.md create mode 100644 papers/items/2026-2605-28607-adaptive-multimodal-agents-based-framework-for-automatic-workflow-execution.md create mode 100644 papers/items/2026-2605-28617-lacuna-safe-agents-as-recursive-program-holes.md create mode 100644 papers/items/2026-2605-28787-do-agents-need-semantic-metadata-a-comparative-study-in-agentic-data-retrieval.md create mode 100644 papers/items/2026-2605-28835-genesisfunc-multi-agent-data-generation-for-accurate-and-generalizable-function-.md create mode 100644 papers/items/2026-2605-28850-representation-signatures-and-risk-feedback-alignment-in-llm-trading-agents.md create mode 100644 papers/items/2026-2605-29341-worldmemarena-evaluating-multimodal-agent-memory-through-action-world-interactio.md create mode 100644 papers/items/2026-2605-29630-entity-collision-a-stratified-protocol-for-attributing-retrieval-lift-in-agent-m.md create mode 100644 papers/items/2026-2605-29640-vikingmem-a-memory-base-management-system-for-stateful-llm-based-applications.md create mode 100644 papers/items/2026-2605-29653-ptcg-bench-can-llm-agents-master-pok-mon-trading-card-game.md create mode 100644 papers/items/2026-2605-29676-notation-matters-a-benchmark-study-of-token-optimized-formats-in-agentic-ai-syst.md create mode 100644 papers/items/2026-2605-29790-evolve-as-a-team-collaborative-self-evolution-for-llm-based-multi-agent-systems.md create mode 100644 papers/items/2026-2605-29801-agentdog-1-5-a-lightweight-and-scalable-alignment-framework-for-ai-agent-safety-.md create mode 100644 papers/items/2026-2605-29861-towards-verifiable-multimodal-deep-research-a-multi-agent-harness-for-interleave.md create mode 100644 papers/items/2026-2605-29960-hijacking-agent-memory-stealthy-trojan-attacks-through-conversational-interactio.md create mode 100644 papers/items/2026-2605-30058-heart-bench-do-llm-agents-exhibit-human-like-psychology.md create mode 100644 papers/items/2026-2605-30090-directorbench-diagnosing-long-form-video-generation-with-personalized-multi-agen.md create mode 100644 papers/items/2026-2605-30407-exploring-autonomous-agentic-data-engineering-for-model-specialization.md create mode 100644 papers/items/2026-2605-30604-an-organization-scoped-llm-agent-runtime-architecture-for-regulated-cybersecurit.md create mode 100644 papers/items/2026-2605-30690-elasticmem-latent-memory-as-a-learnable-resource-for-llm-agents.md create mode 100644 papers/items/2026-2605-30711-sage-a-novelty-gate-for-efficient-memory-evolution-in-agentic-llms.md create mode 100644 papers/items/2026-2605-30858-forecastcompass-guiding-agentic-forecasting-with-adaptive-factor-memory.md create mode 100644 papers/items/2026-2605-30883-trace-task-aware-adaptive-self-evolving-agentic-jailbreaking.md create mode 100644 papers/items/2026-2605-30907-bluefin-benchmarking-llm-agents-on-financial-spreadsheets.md create mode 100644 papers/items/2026-2605-30947-extending-ai-for-research-to-the-humanities-a-multi-agent-framework-for-evidence.md create mode 100644 papers/items/2026-2605-31075-task-focused-memorization-for-multimodal-agents.md create mode 100644 papers/items/2026-2605-31268-mellum2-technical-report.md create mode 100644 papers/items/2026-2605-31278-industrializing-prediction-powered-inference-the-glide-library-for-reliable-gena.md create mode 100644 papers/items/2026-2605-31308-tracegraph-shared-decision-landscapes-for-diagnosing-and-improving-agent-traject.md create mode 100644 papers/items/2026-2605-31377-dynatree-dynamic-agentic-retrieval-tree-for-time-sensitive-news-retrieval.md create mode 100644 papers/items/2026-2606-00198-bagen-are-llm-agents-budget-aware.md create mode 100644 papers/items/2026-2606-00341-rogue-misaligned-agent-behavior-arising-from-ordinary-computer-use.md create mode 100644 papers/items/2026-2606-00610-memgraphrag-memory-based-multi-agent-system-for-graph-retrieval-augmented-genera.md create mode 100644 papers/items/2026-2606-00611-trace-trajectory-risk-aware-compression-for-long-horizon-agent-safety.md create mode 100644 papers/items/2026-2606-00619-mempro-agentic-memory-systems-as-evolvable-programs.md create mode 100644 papers/items/2026-2606-00644-foresci-evaluating-llm-agents-for-forward-looking-ai-research-judgment.md create mode 100644 papers/items/2026-2606-00756-comic-collaborative-memory-and-insights-circulation-for-long-horizon-llm-agents-.md create mode 100644 papers/items/2026-2606-00914-adversarial-feeds-steer-llm-agent-decisions-against-their-defaults.md create mode 100644 papers/items/2026-2606-00915-autonomous-agentic-design-for-photonics.md create mode 100644 papers/items/2026-2606-00922-a-machine-to-machine-knowledge-guided-llm-agent-for-generalizable-radiotherapy-t.md create mode 100644 papers/items/2026-2606-00939-fincom-a-financial-multi-agent-demo-with-disagree-or-commit-deliberation.md create mode 100644 papers/items/2026-2606-01041-expweaver-llm-agents-learn-from-experience-via-latent-rag.md create mode 100644 papers/items/2026-2606-01138-memorywire-a-vendor-neutral-wire-format-for-agent-memory-operations.md create mode 100644 papers/items/2026-2606-01166-braveguard-from-open-world-threats-to-safer-computer-use-agents.md create mode 100644 papers/items/2026-2606-01185-skill-issues-data-centric-optimization-of-lakehouse-agents.md create mode 100644 papers/items/2026-2606-01199-can-llm-agents-sustain-long-horizon-organizational-dynamics.md create mode 100644 papers/items/2026-2606-01222-rag-driven-multi-agent-llm-framework-with-task-decomposition-for-beyond-5g-auto-.md create mode 100644 papers/items/2026-2606-01385-bridging-requirements-and-architecture-multi-agent-orchestration-with-external-k.md create mode 100644 papers/items/2026-2606-01416-self-healing-agentic-orchestrators-for-reliable-tool-augmented-large-language-mo.md create mode 100644 papers/items/2026-2606-01528-joint-agent-memory-and-exploration-learning-via-novelty-signals.md create mode 100644 papers/items/2026-2606-01613-techrag-evidence-gated-multimodal-agentic-rag-for-technical-literature-reasoning.md create mode 100644 papers/items/2026-2606-01815-crab-bench-evaluating-llm-agents-under-complex-task-dependencies-and-human-align.md create mode 100644 papers/items/2026-2606-01961-automedbench-towards-medical-autoresearch-with-agentic-ai-models.md create mode 100644 papers/items/2026-2606-02109-badger-bridging-agentic-and-deterministic-evaluation-for-generative-enterprise-r.md create mode 100644 papers/items/2026-2606-02302-seclaw-spec-driven-security-task-synthesis-for-evaluating-autonomous-agents.md create mode 100644 papers/items/2026-2606-02372-comap-co-evolving-world-models-and-agent-policies-for-llm-agents.md create mode 100644 papers/items/2026-2606-02380-spade-bench-evaluating-spontaneous-strategic-deception-in-agents-via-plan-action.md create mode 100644 papers/items/2026-2606-02388-policy-and-world-modeling-co-training-for-language-agents.md create mode 100644 papers/items/2026-2606-02404-k-browsecomp-a-web-browsing-agent-benchmark-grounded-in-korean-contexts.md create mode 100644 papers/items/2026-2606-02461-agentcl-toward-rigorous-evaluation-of-continual-learning-in-language-agents.md create mode 100644 papers/items/2026-2606-02497-bridging-the-last-mile-of-time-series-forecasting-with-llm-agents.md create mode 100644 papers/items/2026-2606-02812-traj-evolve-a-self-evolving-multi-agent-system-for-patient-trajectory-modeling-i.md create mode 100644 papers/items/2026-2606-02965-what-benchmarks-don-t-measure-the-case-for-evaluating-abstention-competence-in-a.md create mode 100644 papers/items/2026-2606-03108-evotrainer-co-evolving-llm-policies-and-training-harnesses-for-autonomous-agenti.md create mode 100644 papers/items/2026-2606-03135-uncertainty-aware-clarification-in-llm-agents-with-information-gain.md create mode 100644 papers/items/2026-2606-03157-clinicalmc-a-benchmark-for-multi-course-clinical-decision-making-with-large-lang.md create mode 100644 papers/items/2026-2606-03197-memtrain-self-supervised-context-memory-training.md create mode 100644 papers/items/2026-2606-03329-infomem-training-long-context-memory-agents-with-answer-conditioned-information-.md create mode 100644 papers/items/2026-2606-03374-emem-a-hybrid-spatio-temporal-memory-system-for-embodied-agents.md create mode 100644 papers/items/2026-2606-03544-sage-a-quantitative-evaluation-of-socialized-evolution-in-agent-ecosystems.md create mode 100644 papers/items/2026-2606-03657-diagnosing-knowledge-gaps-in-llm-tool-use-an-agentic-benchmark-for-novel-api-acq.md create mode 100644 papers/items/2026-2606-03895-agent-libos-a-runtime-substrate-for-capability-controlled-self-evolving-llm-agen.md create mode 100644 papers/items/2026-2606-04051-rubas-rubric-based-reinforcement-learning-for-agent-safety.md create mode 100644 papers/items/2026-2606-04120-salimory-orchestrating-cognitive-memory-for-conversational-agents.md create mode 100644 papers/items/2026-2606-04296-the-saturation-trap-and-the-subjectivity-of-intervention-timing-why-affect-based.md create mode 100644 papers/items/2026-2606-04315-exploring-cross-scenario-generality-of-agentic-memory-systems-diagnostics-and-a-.md create mode 100644 papers/items/2026-2606-04555-temporal-order-matters-for-agentic-memory-segment-trees-for-long-horizon-agents.md create mode 100644 papers/items/2026-2606-04599-plan-first-judge-later-run-better-a-dmaic-inspired-agentic-system-for-industrial.md create mode 100644 papers/items/2026-2606-04628-rampart-registry-based-agentic-memory-with-priority-aware-runtime-transformation.md create mode 100644 papers/items/2026-2606-04780-personatree-structured-lifecycle-memory-for-person-understanding-in-llm-agents.md create mode 100644 papers/items/2026-2606-04874-agent-planning-benchmark-a-diagnostic-framework-for-planning-capabilities-in-llm.md create mode 100644 papers/items/2026-2606-04990-from-agent-traces-to-trust-a-survey-of-evidence-tracing-and-execution-provenance.md create mode 100644 papers/items/2026-2606-05241-search-time-contamination-in-deep-research-agents-measuring-performance-inflatio.md create mode 100644 papers/items/2026-2606-05263-policy-conditioned-counterfactual-credit-for-verifiable-reinforcement-learning-o.md create mode 100644 papers/items/2026-2606-05414-when-evidence-is-sparse-weakly-supervised-early-failure-alerting-in-dialogs-and-.md create mode 100644 papers/items/2026-2606-05436-ten-headache-specialists-versus-artificial-intelligence-for-clinical-literature-.md create mode 100644 papers/items/2026-2606-05463-psebench-a-controllable-and-verifiable-benchmark-for-evaluating-llms-in-patient-.md create mode 100644 papers/items/2026-2606-05548-adk-arena-evaluating-agent-development-kits-via-llm-as-a-developer.md create mode 100644 papers/items/2026-2606-05558-autoregressive-diffusion-world-models-for-off-policy-evaluation-of-llm-agents.md create mode 100644 papers/items/2026-2606-05622-adaplanbench-evaluating-adaptive-planning-in-large-language-model-agents-under-w.md create mode 100644 papers/items/2026-2606-05658-agent-orchestrated-adaptive-rag-a-comparative-study-on-structured-and-multi-hop-.md create mode 100644 papers/items/2026-2606-05684-adamem-test-time-adaptive-memory-for-language-agents.md create mode 100644 papers/items/2026-2606-05711-beyond-tokens-a-unified-framework-for-latent-communication-in-llm-based-multi-ag.md create mode 100644 papers/items/2026-2606-05805-from-risk-classification-to-action-plan-remediation-a-guardrail-feedback-driven-.md create mode 100644 papers/items/2026-2606-06054-beyond-similarity-trustworthy-memory-search-for-personal-ai-agents.md create mode 100644 papers/items/2026-2606-06090-beyond-semantic-organization-memory-as-execution-state-management-for-long-horiz.md create mode 100644 papers/items/2026-2606-06388-humans-almanac-a-human-collaboration-dataset-of-action-level-mental-model-annota.md create mode 100644 papers/items/2026-2606-06399-collabsim-a-cscw-grounded-methodology-for-investigating-collaborative-competence.md create mode 100644 papers/items/2026-2606-06448-agent-memory-characterization-and-system-implications-of-stateful-long-horizon-w.md create mode 100644 papers/items/2026-2606-06462-benchmark-everything-everywhere-all-at-once.md create mode 100644 papers/items/2026-2606-06473-mlevolve-a-self-evolving-framework-for-automated-machine-learning-algorithm-disc.md create mode 100644 papers/items/2026-2606-07314-qbuglm-an-agentic-benchmarking-framework-for-llm-based-quantum-software-debuggin.md create mode 100644 papers/items/2026-2606-07379-do-coding-agents-deceive-us-detecting-and-preventing-cheating-via-capped-evaluat.md create mode 100644 papers/items/2026-2606-07402-m-3-exam-benchmarking-multimodal-memory-for-realistic-user-agent-interactions.md create mode 100644 papers/items/2026-2606-07591-researchclawbench-a-benchmark-for-end-to-end-autonomous-scientific-research.md create mode 100644 papers/items/2026-2606-07595-visualleakbench-reproducible-action-boundary-propagation-failures-in-vision-lang.md create mode 100644 papers/items/2026-2606-07682-swe-marathon-can-agents-autonomously-complete-ultra-long-horizon-software-work.md create mode 100644 papers/items/2026-2606-07711-rosetta-memory-adaptive-memory-for-cross-llm-agents.md create mode 100644 papers/items/2026-2606-07836-agentic-multi-fidelity-learning-of-quasiparticle-and-excitonic-properties.md create mode 100644 papers/items/2026-2606-07867-the-cold-start-safety-gap-in-llm-agents.md create mode 100644 papers/items/2026-2606-08162-silent-failure-in-llm-agent-systems-the-entropy-principle-and-the-inevitable-dis.md create mode 100644 papers/items/2026-2606-08172-the-governance-of-human-llm-interaction-safety-gating-civility-steering-and-affe.md create mode 100644 papers/items/2026-2606-08274-toward-human-centered-multi-agent-systems-integrating-cognition-culture-values-a.md create mode 100644 papers/items/2026-2606-08340-benchmarking-open-ended-multi-agent-coordination-in-language-agents.md create mode 100644 papers/items/2026-2606-08531-vesta-a-fully-automated-scenario-generation-and-safety-evaluation-framework-for-.md create mode 100644 papers/items/2026-2606-08625-from-holistic-evaluation-to-structured-criteria-rubrics-across-the-evolving-llm-.md create mode 100644 papers/items/2026-2606-08790-rails-verification-native-clearing-for-agentic-commerce.md create mode 100644 papers/items/2026-2606-08960-hardening-agent-benchmarks-with-adversarial-hacker-fixer-loops.md create mode 100644 papers/items/2026-2606-09037-a-multi-agent-system-for-ipmsm-design-optimization-via-an-fea-ai-hybrid-approach.md create mode 100644 papers/items/2026-2606-09071-reflect-intervention-supported-error-attribution-for-silent-failures-in-llm-agen.md create mode 100644 papers/items/2026-2606-09198-mass-deep-research-for-social-sciences-with-memory-augmented-social-simulation.md create mode 100644 papers/items/2026-2606-09316-anything2skill-compiling-external-knowledge-into-reusable-skills-for-agents.md create mode 100644 papers/items/2026-2606-09399-runagent-superbrowser-a-theory-of-autonomous-web-navigation-grounded-in-human-br.md create mode 100644 papers/items/2026-2606-09426-weavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybr.md create mode 100644 papers/items/2026-2606-09447-aliyunconsoleagent-training-web-agents-in-real-world-cloud-environments-via-dist.md create mode 100644 papers/items/2026-2606-09483-memory-beyond-recall-a-dual-process-cognitive-memory-system-for-self-evolving-ll.md create mode 100644 papers/items/2026-2606-09549-secureclaw-clawing-back-control-of-llm-agents.md create mode 100644 papers/items/2026-2606-09738-hdsl-a-hierarchical-domain-specific-language-for-structured-3d-indoor-scene-gene.md create mode 100644 papers/items/2026-2606-09764-iosworld-a-benchmark-for-personally-intelligent-phone-agents.md create mode 100644 papers/items/2026-2606-09774-auto-configuring-scientific-simulators-with-lightweight-coding-agent-adapters.md create mode 100644 papers/items/2026-2606-09863-from-confident-closing-to-silent-failure-characterizing-false-success-in-llm-age.md create mode 100644 papers/items/2026-2606-09961-3spo-state-score-supervised-policy-optimization-for-llm-agents.md create mode 100644 papers/items/2026-2606-10209-less-context-better-agents-efficient-context-engineering-for-long-horizon-tool-u.md create mode 100644 papers/items/2026-2606-10304-mirage-a-polarity-flipping-encoding-subspace-in-llm-agents.md create mode 100644 papers/items/2026-2606-10316-tabclaw-an-interactive-and-self-evolving-agent-for-spreadsheet-manipulation-and-.md create mode 100644 papers/items/2026-2606-10381-agentic-hybrid-rag-for-evidence-grounded-muon-collider-analysis.md create mode 100644 papers/items/2026-2606-10394-stage-claw-automated-state-based-agent-benchmarking-for-realistic-scenarios.md create mode 100644 papers/items/2026-2606-10423-webchallenger-a-reliable-and-efficient-generalist-web-agent.md create mode 100644 papers/items/2026-2606-10507-hipif-hierarchical-planning-and-information-folding-for-long-horizon-llm-agent-l.md create mode 100644 papers/items/2026-2606-10532-activemem-distributed-active-memory-for-long-horizon-llm-reasoning.md create mode 100644 papers/items/2026-2606-10577-agenticnav-zero-shot-vision-and-language-navigation-as-a-tool-calling-harness.md create mode 100644 papers/items/2026-2606-10616-learning-what-to-remember-observability-safe-memory-retention-via-constrained-op.md create mode 100644 papers/items/2026-2606-10677-infini-memory-maintainable-topic-documents-for-long-term-llm-agent-memory.md create mode 100644 papers/items/2026-2606-10684-divide-and-cooperate-role-decomposed-multi-agent-llm-training-with-cross-agent-l.md create mode 100644 papers/items/2026-2606-10742-memvenom-triggered-poisoning-of-multimodal-memories-in-web-agents.md create mode 100644 papers/items/2026-2606-10749-toward-secure-llm-agents-threat-surfaces-attacks-defenses-and-evaluation.md create mode 100644 papers/items/2026-2606-10921-trace-only-what-you-need-structure-aware-on-demand-hypergraph-memory-for-long-do.md create mode 100644 papers/items/2026-2606-10933-frontier-coding-agents-use-metaprogramming-to-adapt-to-unfamiliar-programming-la.md create mode 100644 papers/items/2026-2606-11042-workflow-gym-towards-long-horizon-evaluation-of-computer-use-agentic-tasks-in-re.md create mode 100644 papers/items/2026-2606-11078-a-history-aware-visually-grounded-critic-for-computer-use-agents.md create mode 100644 papers/items/2026-2606-11079-vista-a-versatile-interactive-user-simulation-toolkit-for-agent-evaluation.md create mode 100644 papers/items/2026-2606-11119-trace-a-unified-rollout-budget-allocation-framework-for-efficient-agentic-reinfo.md create mode 100644 papers/items/2026-2606-11176-data-journalist-agent-transforming-data-into-verifiable-multimodal-stories.md create mode 100644 papers/items/2026-2606-11349-knowing-when-to-ask-self-gated-clarification-for-hierarchical-language-agents.md create mode 100644 papers/items/2026-2606-11354-a-zero-shot-multi-agent-framework-for-human-building-interaction-via-programmati.md create mode 100644 papers/items/2026-2606-11680-organize-then-retrieve-hierarchical-memory-navigation-for-efficient-agents.md create mode 100644 papers/items/2026-2606-11688-goal-autopilot-a-verifiable-anti-fabrication-firewall-for-unattended-long-horizo.md create mode 100644 papers/items/2026-2606-11702-medcta-a-benchmark-for-clinical-tool-agents.md create mode 100644 papers/items/2026-2606-11869-agents-all-the-way-down-a-methodology-for-building-custom-ai-agents-from-substra.md create mode 100644 papers/items/2026-2606-12195-internvideo3-agentify-foundation-models-with-multimodal-contextual-reasoning.md create mode 100644 papers/items/2026-2606-12320-a-five-plane-reference-architecture-for-runtime-governance-of-production-ai-agen.md create mode 100644 papers/items/2026-2606-12341-ocelot-inference-leakage-budgets-for-privacy-preserving-llm-agents.md create mode 100644 papers/items/2026-2606-12344-claw-swe-bench-a-benchmark-for-evaluating-openclaw-style-agent-harnesses-on-codi.md create mode 100644 papers/items/2026-2606-12384-appo-agentic-procedural-policy-optimization.md create mode 100644 papers/items/2026-2606-12563-arbor-tree-search-as-a-cognition-layer-for-autonomous-agents.md create mode 100644 papers/items/2026-2606-12586-beyond-attack-success-rate-examining-trigger-leakage-in-vision-language-agentic-.md create mode 100644 papers/items/2026-2606-12634-keep-policy-gradient-in-charge-sibling-guided-credit-distillation-for-long-horiz.md create mode 100644 papers/items/2026-2606-12657-trajgenagent-a-hierarchical-llm-agent-for-human-mobility-trajectory-generation.md create mode 100644 papers/items/2026-2606-12674-evoflux-inference-time-evolution-of-executable-tool-workflows-for-compact-agents.md create mode 100644 papers/items/2026-2606-12703-smsr-certified-defence-against-runtime-memory-poisoning-in-persistent-llm-agent-.md create mode 100644 papers/items/2026-2606-12780-proplay-procedural-world-models-for-self-evolving-llm-agents.md create mode 100644 papers/items/2026-2606-12837-lohosearch-benchmarking-long-horizon-search-agents-beyond-the-human-difficulty-c.md create mode 100644 papers/items/2026-2606-12945-learning-what-to-remember-a-cognitively-grounded-multi-factor-value-model-for-ag.md create mode 100644 papers/items/2026-2606-13148-terrabench-can-agents-reason-over-heterogeneous-earth-system-data.md create mode 100644 papers/items/2026-2606-13177-memrefine-llm-guided-compression-for-long-term-agent-memory.md create mode 100644 papers/items/2026-2606-13192-reasoning-for-mobile-user-experience-with-multimodal-llms-task-benchmark-and-app.md create mode 100644 papers/items/2026-2606-13317-skillcat-contrastive-assessment-and-topology-aware-skill-self-evolution-for-llm-.md create mode 100644 papers/items/2026-2606-13385-who-pays-the-price-stakeholder-centric-prompt-injection-benchmarking-for-real-wo.md create mode 100644 papers/items/2026-2606-13602-epibench-verifiable-evaluation-of-ai-agents-on-epigenomics-analysis.md create mode 100644 papers/items/2026-2606-13608-agentbeats-agentifying-agent-assessment-for-openness-standardization-and-reprodu.md create mode 100644 papers/items/2026-2606-13643-recursive-agent-harnesses.md create mode 100644 papers/items/2026-2606-13663-hypertool-beyond-step-wise-tool-calls-for-tool-augmented-agents.md create mode 100644 papers/items/2026-2606-13686-benchmarking-web-agent-safety-under-e-commerce-deceptive-interfaces.md create mode 100644 papers/items/2026-2606-13904-sana-what-matters-for-qa-agents-over-massive-data-lakes.md create mode 100644 papers/items/2026-2606-13994-hidden-in-plain-sight-benchmarking-agent-safety-against-decomposition-attacks-wi.md create mode 100644 papers/items/2026-2606-14106-naive-visual-memory-is-not-enough-a-failure-mode-study-of-gui-agents.md create mode 100644 papers/items/2026-2606-14470-gitofthoughts-version-controlled-reasoning-and-agent-memory-you-can-replay-diff-.md create mode 100644 papers/items/2026-2606-14502-from-chatbot-to-digital-colleague-the-paradigm-shift-toward-persistent-autonomou.md create mode 100644 papers/items/2026-2606-14517-from-shield-to-target-denial-of-service-attacks-on-llm-based-agent-guardrails.md create mode 100644 papers/items/2026-2606-14571-streammembench-streaming-evaluation-of-agent-memory-for-future-oriented-assistan.md create mode 100644 papers/items/2026-2606-14574-simmer-benchmarking-latent-failures-in-llm-executable-planning-with-a-world-mode.md create mode 100644 papers/items/2026-2606-14790-xflow-an-executable-protocol-programming-system-for-reliable-multi-agent-workflo.md create mode 100644 papers/items/2026-2606-14805-knowledge-based-zero-replay-debugging-of-multi-agent-llm-traces.md create mode 100644 papers/items/2026-2606-15017-are-online-skill-and-memory-modules-always-worth-their-tokens-a-budget-constrain.md create mode 100644 papers/items/2026-2606-15034-osguard-a-benchmark-for-safety-in-computer-use-agents.md create mode 100644 papers/items/2026-2606-15079-ling-and-ring-2-6-technical-report-efficient-and-instant-agentic-intelligence-at.md create mode 100644 papers/items/2026-2606-15152-can-agents-read-the-room-benchmarking-visual-social-intelligence-in-multimodal-s.md create mode 100644 papers/items/2026-2606-15242-benign-in-isolation-harmful-in-composition-security-risks-in-agent-skill-ecosyst.md create mode 100644 papers/items/2026-2606-15376-coagent-concurrency-control-for-multi-agent-systems.md create mode 100644 papers/items/2026-2606-15591-agentic-retrieval-and-reinforcement-learned-equation-chains-a-controlled-generat.md create mode 100644 papers/items/2026-2606-15609-fragfuse-bypassing-access-control-of-large-language-model-agents-via-memory-base.md create mode 100644 papers/items/2026-2606-15684-multi-agent-framework-for-time-sensitive-complementary-collaboration-in-minecraf.md create mode 100644 papers/items/2026-2606-15709-ai-driven-framework-for-adaptive-water-network-management-with-proof-of-concept-.md create mode 100644 papers/items/2026-2606-15862-retailbench-benchmarking-long-horizon-reasoning-and-coherent-decision-making-of-.md create mode 100644 papers/items/2026-2606-15874-llm-as-code-agentic-programming-for-agent-harness.md create mode 100644 papers/items/2026-2606-15903-control-plane-placement-shapes-forgetting-an-architectural-study-of-agent-memory.md create mode 100644 papers/items/2026-2606-15906-mage-rag-multigranular-adaptive-graph-evidence-for-agentic-multimodal-rag-in-lon.md create mode 100644 papers/items/2026-2606-15931-deeproot-a-kg-coordinated-multi-agent-system-for-therapeutic-reasoning-over-hist.md create mode 100644 papers/items/2026-2606-15994-agentic-framework-for-deep-learning-workload-migration-via-in-context-learning.md create mode 100644 papers/items/2026-2606-16111-towards-pareto-optimal-tool-integrated-agents-with-pareto-ranking-policy-optimiz.md create mode 100644 papers/items/2026-2606-16295-visualclaw-a-real-time-personalized-agent-for-the-physical-world.md create mode 100644 papers/items/2026-2606-16420-transferable-self-evolving-playbooks-for-agentic-security-auditing.md create mode 100644 papers/items/2026-2606-16432-accord-action-conditioned-contextual-grounding-for-language-agents.md create mode 100644 papers/items/2026-2606-16481-steering-emotional-dynamics-for-art-therapy-controllable-narrative-script-genera.md create mode 100644 papers/items/2026-2606-16534-generated-parallel-scalable-a-study-of-agentic-ai-generated-julia-code-on-superc.md create mode 100644 papers/items/2026-2606-16576-can-llm-agents-infer-world-models-evidence-from-agentic-automata-learning.md create mode 100644 papers/items/2026-2606-16591-sing-synthetic-intention-graph-for-scalable-active-tool-discovery-in-llm-agents.md create mode 100644 papers/items/2026-2606-16613-coffeebench-benchmarking-long-horizon-llm-agents-in-heterogeneous-multi-agent-ec.md create mode 100644 papers/items/2026-2606-16659-fraudsmswalker-benchmarking-agentic-large-language-models-for-sms-to-webpage-fra.md create mode 100644 papers/items/2026-2606-16748-mypcbench-a-benchmark-for-personally-intelligent-computer-use-agents.md create mode 100644 papers/items/2026-2606-16774-openclaw-skill-collective-skill-tree-search-for-agentic-large-language-models.md create mode 100644 papers/items/2026-2606-16802-labosbench-benchmarking-computer-use-agents-for-scientific-instrument-control.md create mode 100644 papers/items/2026-2606-16813-gist-cmtf-goal-state-inference-for-causal-minimal-tool-filtering-in-llm-agents.md create mode 100644 papers/items/2026-2606-16839-towards-llm-accelerated-rapid-reviews-for-software-tool-discovery-case-for-log-a.md create mode 100644 papers/items/2026-2606-16871-human-on-the-bridge-scalable-evaluation-for-ai-agents.md create mode 100644 papers/items/2026-2606-17041-benchmarking-llm-agents-on-meta-analysis-articles-from-nature-portfolio.md create mode 100644 papers/items/2026-2606-17076-cmip-forge-an-agentic-system-that-retrieves-computes-and-self-reviews-climate-sc.md create mode 100644 papers/items/2026-2606-17114-an-evaluation-of-data-leakage-risks-in-tool-using-llm-agents-in-realistic-scenar.md create mode 100644 papers/items/2026-2606-17246-geodisaster-benchmarking-orchestrated-agents-for-operational-disaster-geo-intell.md create mode 100644 papers/items/2026-2606-17368-distributed-general-purpose-agent-networks-architecture-key-mechanisms-and-proto.md create mode 100644 papers/items/2026-2606-17383-model-validation-of-agentic-ai-systems-a-pomdp-based-framework-for-belief-state-.md create mode 100644 papers/items/2026-2606-17449-mode-rag-manifold-outlier-diagnosis-and-energy-based-retrieval-augmented-generat.md create mode 100644 papers/items/2026-2606-17453-mapsatisfybench-benchmarking-satisfaction-aware-map-agents-through-behavior-grou.md create mode 100644 papers/items/2026-2606-17459-can-llms-be-ceos-benchmarking-strategic-resource-reallocation-with-multi-role-ag.md create mode 100644 papers/items/2026-2606-17573-cordon-semantic-transactions-for-tool-using-llm-agents.md create mode 100644 papers/items/2026-2606-17680-envrl-learn-from-environment-dynamics-in-agentic-reinforcement-learning.md create mode 100644 papers/items/2026-2606-18023-loopcoder-v2-only-loop-once-for-efficient-test-time-computation-scaling.md create mode 100644 papers/items/2026-2606-18037-provenanceguard-source-aware-factuality-verification-for-mcp-based-llm-agents.md create mode 100644 papers/items/2026-2606-18051-compositional-skill-routing-for-llm-agents-decompose-retrieve-and-compose.md create mode 100644 papers/items/2026-2606-18068-agentic-ai-based-framework-for-mitigating-premature-diagnostic-handoff-and-silen.md create mode 100644 papers/items/2026-2606-18142-your-ai-travel-agent-would-book-you-a-bullfight-an-agentic-benchmark-for-implici.md create mode 100644 papers/items/2026-2606-18272-mitigating-anchoring-bias-in-llm-based-agents-for-energy-efficient-6g-autonomous.md create mode 100644 papers/items/2026-2606-18356-safeclawbench-separating-semantic-audit-evidence-and-sandbox-harm-in-tool-using-.md create mode 100644 papers/items/2026-2606-18363-guava-an-effective-and-universal-harness-for-embodied-manipulation.md create mode 100644 papers/items/2026-2606-18406-coremem-riemannian-retrieval-and-fisher-guided-distillation-for-long-term-memory.md create mode 100644 papers/items/2026-2606-18467-toolchain-crc-conformal-risk-control-for-agentic-ai-under-retrieval-and-tool-use.md create mode 100644 papers/items/2026-2606-18502-towards-scalable-customization-and-deployment-of-multi-agent-systems-for-enterpr.md create mode 100644 papers/items/2026-2606-18619-code-augur-agentic-vulnerability-detection-via-specification-inference.md create mode 100644 papers/items/2026-2606-18671-hansel-extracting-breadcrumbs-from-web-agent-trajectories-for-interactive-verifi.md create mode 100644 papers/items/2026-2606-18789-poweragentbench-ss-a-benchmark-for-agentic-ai-in-power-system-steady-state-studi.md create mode 100644 papers/items/2026-2606-18829-gatemem-benchmarking-memory-governance-in-multi-principal-shared-memory-agents.md create mode 100644 papers/items/2026-2606-18950-rtsgamebench-an-rts-benchmark-for-strategic-reasoning-by-vision-language-models.md create mode 100644 papers/items/2026-2606-19063-pypiline-malicious-pypi-package-detection-via-suspicious-api-knowledge-and-agent.md create mode 100644 papers/items/2026-2606-19242-runtime-compliance-verification-for-ai-agents.md create mode 100644 papers/items/2026-2606-19245-txbench-pp-analyzing-ai-agent-performance-on-small-molecule-preclinical-pharmaco.md create mode 100644 papers/items/2026-2606-19409-openrath-session-centered-runtime-state-for-agent-systems.md create mode 100644 papers/items/2026-2606-19464-deontic-policies-for-runtime-governance-of-agentic-ai-systems.md create mode 100644 papers/items/2026-2606-19613-staminabench-stress-testing-coding-agents-over-100-interaction-turns.md create mode 100644 papers/items/2026-2606-19704-beyond-static-leaderboards-predictive-validity-for-the-evaluation-of-llm-agents.md create mode 100644 papers/items/2026-2606-19787-oragentbench-can-llm-agents-solve-challenging-operations-research-tasks-end-to-e.md create mode 100644 papers/items/2026-2606-19812-human-on-the-loop-orchestration-for-ai-assisted-legal-discovery.md create mode 100644 papers/items/2026-2606-19852-prompt-plan-extract-zero-shot-agentic-llms-workflows-for-lung-pathology-extracti.md create mode 100644 papers/items/2026-2606-19899-measuring-biological-capabilities-and-risks-of-ai-agents.md create mode 100644 papers/items/2026-2606-19926-memgui-agent-an-end-to-end-long-horizon-mobile-gui-agent-with-proactive-context-.md create mode 100644 papers/items/2026-2606-19930-mobileforge-annotation-free-adaptation-for-mobile-gui-agents-with-hierarchical-f.md create mode 100644 papers/items/2026-2606-19980-enpire-agentic-robot-policy-self-improvement-in-the-real-world.md create mode 100644 papers/items/2026-2606-20023-when-lower-privileges-suffice-investigating-over-privileged-tool-selection-in-ll.md create mode 100644 papers/items/2026-2606-20041-ai-economist-agent-an-agentic-framework-for-model-grounded-economic-analysis-wit.md create mode 100644 papers/items/2026-2606-20047-pacms-submodular-context-selection-as-a-pluggable-engine-for-llm-agents.md create mode 100644 papers/items/2026-2606-20243-phoenix-safe-github-issue-resolution-via-multi-agent-llms.md create mode 100644 papers/items/2026-2606-20401-poweragentbench-dyn-a-benchmark-for-agentic-ai-in-power-system-dynamic-studies.md create mode 100644 papers/items/2026-2606-20470-analyzing-defensive-misdirection-against-model-guided-automated-attacks-on-agent.md create mode 100644 papers/items/2026-2606-20479-groundcontrol-anticipating-navigation-failures-in-vision-language-agents-via-tra.md create mode 100644 papers/items/2026-2606-20510-efficient-and-sound-probabilistic-verification-for-ai-agents.md create mode 100644 papers/items/2026-2606-20512-probe-and-refine-tuning-of-repository-guidance-for-coding-agents.md create mode 100644 papers/items/2026-2606-20515-s-agent-spatial-tool-use-elicits-reasoning-for-spatial-intelligence.md create mode 100644 papers/items/2026-2606-20573-aona-a-comprehensive-architecture-and-workflow-design-for-global-agentic-collabo.md create mode 100644 papers/items/2026-2606-20629-specialize-roles-mix-deployments-pushing-the-cost-accuracy-frontier-of-llm-agent.md create mode 100644 papers/items/2026-2606-20717-mirage-stealthy-visual-prompt-injection-for-vulnerability-detection-in-web-agent.md create mode 100644 papers/items/2026-2606-20785-fara-1-5-scalable-learning-environments-for-computer-use-agents.md create mode 100644 papers/items/2026-2606-20922-think-twice-before-you-act-protecting-llm-agents-against-tool-description-poison.md create mode 100644 papers/items/2026-2606-20950-power-systems-agent-benchmark-executable-evaluation-of-ai-agents-in-electric-pow.md create mode 100644 papers/items/2026-2606-20954-learning-what-not-to-forget-long-horizon-agent-memory-from-a-few-kilobytes-of-le.md create mode 100644 papers/items/2026-2606-21013-agentic-time-machine-as-an-infrastructure-for-future-event-forecasting.md create mode 100644 papers/items/2026-2606-21123-a-multi-agent-audit-framework-for-high-stakes-reasoning-evaluation-and-interpret.md create mode 100644 papers/items/2026-2606-21129-agenticos-an-intent-oriented-secure-operating-system-architecture-for-autonomous.md create mode 100644 papers/items/2026-2606-21228-sakana-fugu-technical-report.md create mode 100644 papers/items/2026-2606-21401-swarmx-agentic-scheduling-for-low-latency-agentic-systems.md create mode 100644 papers/items/2026-2606-21409-don-t-blindly-trust-it-how-unreliable-feedback-breaks-tool-using-llm-agents.md create mode 100644 papers/items/2026-2606-21445-autoras-learning-robust-agentic-systems-with-primitive-representations.md create mode 100644 papers/items/2026-2606-21553-dissecting-agentic-rag-a-component-ablation-for-multi-hop-qa-with-a-local-7b-mod.md create mode 100644 papers/items/2026-2606-21565-composing-verifiable-conceptual-models-via-building-blocks-towards-design-time-v.md create mode 100644 papers/items/2026-2606-21627-counsel-a-meta-evaluation-dataset-for-agentic-tasks.md create mode 100644 papers/items/2026-2606-21649-evoembedding-evolvable-representations-for-long-context-retrieval-and-agentic-me.md create mode 100644 papers/items/2026-2606-21710-privacyalign-contextual-privacy-alignment-for-llm-agents.md create mode 100644 papers/items/2026-2606-21732-safe-to-check-unsafe-to-use-relinking-at-the-compression-boundary-of-llm-agents.md create mode 100644 papers/items/2026-2606-21740-training-the-orchestrator-a-supervised-approach-to-end-to-end-pddl-planning-with.md create mode 100644 papers/items/2026-2606-21836-agentdse-reasoning-augmented-architectural-design-space-exploration.md create mode 100644 papers/items/2026-2606-21842-agent-assisted-side-channel-attacks-on-non-prefix-kv-cache-in-rag.md create mode 100644 papers/items/2026-2606-21877-agentriskbom-a-risk-scoping-security-bill-of-materials-for-agentic-ai-systems.md create mode 100644 papers/items/2026-2606-21963-holmes-multimodal-agentic-diagnosis-for-mixed-language-mobile-crashes-at-industr.md create mode 100644 papers/items/2026-2606-22030-nous-a-predictive-world-model-for-long-term-agent-memory.md create mode 100644 papers/items/2026-2606-22082-codeteam-an-llm-powered-multi-agent-framework-for-repository-level-code-generati.md create mode 100644 papers/items/2026-2606-22110-traceview-interactive-visualization-of-agentic-program-repair-trajectories.md create mode 100644 papers/items/2026-2606-22151-novelty-aware-agentic-retrieval-comparing-research-contributions-through-structu.md create mode 100644 papers/items/2026-2606-22263-revelio-cost-efficient-agentic-memory-safety-vulnerability-detection-for-reposit.md create mode 100644 papers/items/2026-2606-22330-hypothesis-driven-skill-optimization-for-llm-agents.md create mode 100644 papers/items/2026-2606-22388-planbench-xl-evaluating-long-horizon-planning-of-llm-tool-use-agents-in-large-sc.md create mode 100644 papers/items/2026-2606-22417-code-isn-t-memory-a-structural-codebase-index-inside-a-coding-agent.md create mode 100644 papers/items/2026-2606-22484-governed-ai-assisted-engineering-graduated-human-oversight-for-agentic-code-gene.md create mode 100644 papers/items/2026-2606-22495-grounded-scaling-why-agentic-ai-needs-deterministic-environments.md create mode 100644 papers/items/2026-2606-22557-macagentbench-benchmarking-ai-agents-on-real-world-macos-desktop.md create mode 100644 papers/items/2026-2606-22610-paperclaw-harnessing-agents-for-autonomous-research-and-human-in-the-loop-refine.md create mode 100644 papers/items/2026-2606-22647-raven-agentic-rag-for-automated-vulnerability-repair.md create mode 100644 papers/items/2026-2606-22673-agentlens-interpretable-safety-steering-via-mechanistic-subspaces-for-multi-turn.md create mode 100644 papers/items/2026-2606-22678-rigorbench-benchmarking-engineering-process-discipline-in-autonomous-ai-coding-a.md create mode 100644 papers/items/2026-2606-22737-groundeval-a-deterministic-replacement-for-llm-as-judge-in-stateful-agent-evalua.md create mode 100644 papers/items/2026-2606-22741-grade-graph-representation-of-llm-agent-dependency-and-execution.md create mode 100644 papers/items/2026-2606-22844-ramem-contextual-reinstatement-for-long-term-agentic-memory.md create mode 100644 papers/items/2026-2606-22864-when-auc-0-998-is-not-enough-a-candidate-evaluation-protocol-for-hidden-state-pr.md create mode 100644 papers/items/2026-2606-22948-envs-environment-native-verified-search-for-long-horizon-gui-agents.md create mode 100644 papers/items/2026-2606-22953-plans-don-t-persist-why-context-management-is-load-bearing-for-llm-agents.md create mode 100644 papers/items/2026-2606-23032-ipo-finance-agent-benchmark-of-llm-financial-analysts-beyond-finance-agent-v2-wi.md create mode 100644 papers/items/2026-2606-23130-understanding-the-in-security-of-vibe-coded-applications.md create mode 100644 papers/items/2026-2606-23195-memory-contagion-cross-temporal-propagation-of-evaluator-bias-via-agent-memory.md create mode 100644 papers/items/2026-2606-23283-towards-root-memories-benchmarking-and-enhancing-implicit-logical-memory-retriev.md create mode 100644 papers/items/2026-2606-23327-videoagent-all-in-one-framework-for-video-understanding-and-editing.md create mode 100644 papers/items/2026-2606-23343-group-selection-promotes-prosocial-prompts-in-populations-of-llm-agents.md create mode 100644 papers/items/2026-2606-23565-holoagent-0-a-unified-embodied-agent-framework-with-3d-spatial-memory.md create mode 100644 papers/items/2026-2606-23664-mas-promptbench-when-does-prompt-optimization-improve-multi-agent-llm-systems.md create mode 100644 papers/items/2026-2606-23752-esaa-conversational-an-event-sourced-memory-layer-for-continuity-handoff-and-cur.md create mode 100644 papers/items/2026-2606-23764-emergent-relational-order-in-llm-agent-societies-from-collective-affect-to-autho.md create mode 100644 papers/items/2026-2606-23927-rift-bench-dynamic-red-teaming-for-agentic-ai-systems.md create mode 100644 papers/items/2026-2606-23991-critique-of-agent-model.md create mode 100644 papers/items/2026-2606-24193-skychain-intelligence-a-blockchain-secured-multi-agent-drl-framework-for-low-alt.md create mode 100644 papers/items/2026-2606-24235-sp-mind-an-autonomous-reasoning-agent-for-spatial-proteomics-analysis.md create mode 100644 papers/items/2026-2606-24322-securing-llm-agent-long-term-memory-against-poisoning-non-malleable-origin-bound.md create mode 100644 papers/items/2026-2606-24402-poisoned-playbooks-demystifying-knowledge-poisoning-effects-on-ai-security-agent.md create mode 100644 papers/items/2026-2606-24416-agentic-ai-for-bilevel-long-term-optimization-of-policy-driven-physical-layer-sy.md create mode 100644 papers/items/2026-2606-24437-rem-moa-reasoning-memory-sustains-mixture-of-agents-scaling.md create mode 100644 papers/items/2026-2606-24453-bayesian-control-for-coding-agents.md create mode 100644 papers/items/2026-2606-24515-reinforcement-learning-for-computer-use-agents-with-autonomous-evaluation.md create mode 100644 papers/items/2026-2606-24525-viscritic-visual-state-comparison-as-process-reward-for-gui-agents.md create mode 100644 papers/items/2026-2606-24535-governed-shared-memory-for-multi-agent-llm-systems.md create mode 100644 papers/items/2026-2606-24551-gui-vs-cli-execution-bottlenecks-in-screen-only-and-skill-mediated-computer-use-.md create mode 100644 papers/items/2026-2606-24595-memprobe-probing-long-term-agent-memory-via-hidden-user-state-recovery.md create mode 100644 papers/items/2026-2606-24597-qwen-agentworld-language-world-models-for-general-agents.md create mode 100644 papers/items/2026-2606-24623-privacy-preserving-rag-via-multi-agent-semantic-rewriting-achieving-confidential.md create mode 100644 papers/items/2026-2606-24626-safari-scaling-long-horizon-agentic-fault-attribution-via-active-investigation.md create mode 100644 papers/items/2026-2606-24649-agentic-collaborative-cognition-for-zero-shot-3d-understanding.md create mode 100644 papers/items/2026-2606-24689-automated-summarization-of-software-documents-an-llm-based-multi-agent-approach.md create mode 100644 papers/items/2026-2606-24694-supplynet-supporting-visual-exploratory-learning-in-supply-chain-via-contextual-.md create mode 100644 papers/items/2026-2606-24775-are-we-ready-for-an-agent-native-memory-system.md create mode 100644 papers/items/2026-2606-24779-deepbd-a-grounded-agentic-workflow-for-variant-prioritization-and-diagnosis-of-g.md create mode 100644 papers/items/2026-2606-24820-sherloc-structured-diagnostic-localization-for-code-repair-agents.md create mode 100644 papers/items/2026-2606-24839-grading-the-grader-lessons-from-evaluating-an-agentic-data-analysis-system.md create mode 100644 papers/items/2026-2606-24855-openthoughts-agent-data-recipes-for-agentic-models.md create mode 100644 papers/items/2026-2606-24937-the-hitchhiker-s-guide-to-agentic-ai-from-foundations-to-systems.md create mode 100644 papers/items/2026-2606-24976-diagnosing-and-mitigating-compounding-failures-in-agentic-persuasion-via-taxonom.md create mode 100644 papers/items/2026-2606-25115-forget-to-improve-on-device-llm-agent-continual-learning-via-budget-curated-memo.md create mode 100644 papers/items/2026-2606-25139-buildrix-an-open-platform-for-sharing-and-benchmarking-agentic-ai-skills-in-buil.md create mode 100644 papers/items/2026-2606-25161-trustmem-learning-trustworthy-memory-consolidation-for-llm-agents-with-long-term.md create mode 100644 papers/items/2026-2606-25189-actplane-programmable-os-level-policy-enforcement-for-agent-harnesses.md create mode 100644 papers/items/2026-2606-25191-to-isolate-or-to-score-model-adaptive-assessment-for-cost-efficient-multi-agent-.md create mode 100644 papers/items/2026-2606-25195-sok-ai-secure-code-generation-progress-pitfalls-and-paths-forward.md create mode 100644 papers/items/2026-2606-25206-raven-long-horizon-reasoning-navigation-with-a-visuo-spatio-temporal-memory.md create mode 100644 papers/items/2026-2606-25334-bridging-the-post-discharge-gap-a-traceable-multi-agent-framework-for-safe-and-c.md create mode 100644 papers/items/2026-2606-25358-agentic-knowledge-tracing-a-multi-agent-llm-architecture-for-stealth-assessment-.md create mode 100644 papers/items/2026-2606-25361-memory-makes-the-difference-evaluating-how-different-memory-roles-shape-conversa.md create mode 100644 papers/items/2026-2606-25400-brainagent-a-large-language-model-driven-multi-agent-framework-for-autonomous-br.md create mode 100644 papers/items/2026-2606-25484-from-causal-discovery-to-implementation-an-agentic-ai-framework-for-e-scooter-mo.md create mode 100644 papers/items/2026-2606-25514-unlocking-model-potentials-through-adaptive-multi-agent-scaffolding-for-efficien.md create mode 100644 papers/items/2026-2606-25519-quantization-inflates-reasoning-token-inflation-as-a-hidden-cost-of-low-bit-reas.md create mode 100644 papers/items/2026-2606-25588-intenttester-intent-driven-multi-agent-framework-for-cross-library-test-migratio.md create mode 100644 papers/items/2026-2606-25622-probabilistic-agents-in-deterministic-audits-evaluating-multi-agent-systems-for-.md create mode 100644 papers/items/2026-2606-25651-medguards-multi-agent-system-for-reliable-medical-error-detection-and-correction.md create mode 100644 papers/items/2026-2606-25656-is-graphrag-needed-from-basic-rag-to-graph-agentic-solutions-with-context-optimi.md create mode 100644 papers/items/2026-2606-25705-gui-agent-guided-exploration-of-user-sensitive-screens.md create mode 100644 papers/items/2026-2606-25760-uncertainty-quantification-for-computer-use-agents-a-benchmark-across-vision-lan.md create mode 100644 papers/items/2026-2606-25819-beyond-function-calling-benchmarking-tool-using-agents-under-tool-environment-un.md create mode 100644 papers/items/2026-2606-25899-manipulation-is-task-dependent-a-multi-axis-multi-environment-evaluation-of-fron.md create mode 100644 papers/items/2026-2606-26057-the-unfireable-safety-kernel-execution-time-ai-alignment-for-ai-agents-and-other.md create mode 100644 papers/items/2026-2606-26203-agentic-analysis-for-agentic-infrastructure-an-llm-powered-pipeline-for-comparat.md create mode 100644 papers/items/2026-2606-26205-knowledge-augmented-agentic-ai-for-mental-health-medication-information-seeking.md create mode 100644 papers/items/2026-2606-26216-cyberchainbench-can-ai-agents-secure-smart-contracts-against-real-world-on-chain.md create mode 100644 papers/items/2026-2606-26300-the-verification-horizon-no-silver-bullet-for-coding-agent-rewards.md create mode 100644 papers/items/2026-2606-26346-how-do-tool-augmented-llm-agents-perform-on-real-world-energy-analytics-tasks.md create mode 100644 papers/items/2026-2606-26356-instruction-bleed-cross-module-interference-in-prompt-composed-agentic-systems.md create mode 100644 papers/items/2026-2606-26403-profilefoundry-a-synthetic-person-object-substrate-for-privacy-memory-and-tool-u.md create mode 100644 papers/items/2026-2606-26453-optimizing-cuda-like-a-human-micro-profiling-tools-as-expert-surrogates-for-llm-.md create mode 100644 papers/items/2026-2606-26479-adaptive-evaluation-of-out-of-band-defenses-against-prompt-injection-in-llm-agen.md create mode 100644 papers/items/2026-2606-26511-temporal-validity-in-retrieval-memory-eliminating-stale-fact-errors-for-ai-agent.md create mode 100644 papers/items/2026-2606-26524-vigil-runtime-enforcement-of-behavioral-specifications-in-ai-agent-skills.md create mode 100644 papers/items/2026-2606-26614-hilsva-design-and-evaluation-of-a-human-in-the-loop-agentic-system-for-scientifi.md create mode 100644 papers/items/2026-2606-26627-agents-that-know-too-much-a-data-centric-survey-of-privacy-in-llm-agents.md create mode 100644 papers/items/2026-2606-26649-autoformalization-of-agent-instructions-into-policy-as-code.md create mode 100644 papers/items/2026-2606-26721-knowledge-based-pull-requests-a-trusted-workflow-for-agent-mediated-knowledge-co.md create mode 100644 papers/items/2026-2606-26758-egg-an-expert-guided-agent-framework-for-kernel-generation.md create mode 100644 papers/items/2026-2606-26793-mirror-novelty-constrained-memory-guided-mcts-red-teaming-for-agentic-rag.md create mode 100644 papers/items/2026-2606-26806-memory-depth-not-memory-access-selective-parametric-consolidation-for-long-runni.md create mode 100644 papers/items/2026-2606-26883-econsimulacra-a-digital-twin-platform-of-socio-economic-systems-powered-by-llm-a.md create mode 100644 papers/items/2026-2606-26918-diagnosing-task-insensitivity-in-language-agents.md create mode 100644 papers/items/2026-2606-26924-a-deterministic-control-plane-for-llm-coding-agents.md create mode 100644 papers/items/2026-2606-26960-toward-agentic-sysadmin-rethinking-system-administration-with-ai-agents.md create mode 100644 papers/items/2026-2606-27009-semantic-early-stopping-for-iterative-llm-agent-loops.md create mode 100644 papers/items/2026-2606-27154-openrca-2-0-from-outcome-labels-to-causal-process-supervision.md create mode 100644 papers/items/2026-2606-27243-nova-a-verification-aware-agent-harness-for-architecture-evolution-in-industrial.md create mode 100644 papers/items/2026-2606-27330-empowering-gui-agents-via-autonomous-experience-exploration-and-hindsight-experi.md create mode 100644 papers/items/2026-2606-27350-chia-an-open-source-framework-for-principled-agentic-ai-driven-hardware-software.md create mode 100644 papers/items/2026-2606-27397-sidconarena-an-environment-evaluating-agents-in-open-ended-positive-sum-bargaini.md create mode 100644 papers/items/2026-2606-27406-towards-evaluation-of-implicit-software-world-models-in-coding-llms.md create mode 100644 papers/items/2026-2606-27416-glite-arf-verifier-driven-research-with-parallel-llm-coding-agents.md create mode 100644 papers/items/2026-2606-27472-supersede-diagnosing-and-training-the-memory-update-gap-in-llm-agents.md create mode 100644 papers/items/2026-2606-27483-internalizing-the-future-a-unified-agentic-training-paradigm-for-world-model-pla.md create mode 100644 papers/items/2026-2606-27492-queenbee-planner-skill-evolving-communication-topologies-for-token-efficient-llm.md create mode 100644 papers/items/2026-2606-27499-dmv-bench-diagnosing-long-horizon-multimodal-agents-visual-memory-with-incidenta.md create mode 100644 papers/items/2026-2606-27632-yuvion-llm-an-adversarially-aware-large-language-model-for-content-and-ai-safety.md create mode 100644 papers/items/2026-2606-27806-agent-vs-parametric-world-models-hybrid-planning-for-reliable-language-agents.md create mode 100644 papers/items/2026-2606-27929-when-multi-robot-systems-meet-agentic-ai-towards-embodied-collective-intelligenc.md create mode 100644 papers/items/2026-2606-27990-advancedshellm-a-stateful-multi-agent-llm-honeypot-for-ssh-deception.md create mode 100644 papers/items/2026-2606-28011-from-detection-to-action-using-llm-agents-for-fault-tolerant-control.md create mode 100644 papers/items/2026-2606-28061-toolprivacybench-benchmarking-purpose-bound-privacy-in-tool-using-llm-agents.md create mode 100644 papers/items/2026-2606-28182-llawco-learning-laws-of-cooperation-for-modeling-embodied-multi-agent-behavior.md create mode 100644 papers/items/2026-2606-28187-gbc-gradient-based-connections-for-optimizing-multi-agent-systems.md create mode 100644 papers/items/2026-2606-28270-agent-native-immune-system-architecture-taxonomy-and-engineering.md create mode 100644 papers/items/2026-2606-28279-agentic-hardware-design-as-repository-level-code-evolution.md create mode 100644 papers/items/2026-2606-28349-hmars-a-hierarchical-multi-agent-memory-system-for-long-context-reasoning.md create mode 100644 papers/items/2026-2606-28360-carolina-guide-a-multi-agent-rag-system-with-institutional-guardrails-for-academ.md create mode 100644 papers/items/2026-2606-28374-recursive-self-evolving-agents-via-held-out-selection.md create mode 100644 papers/items/2026-2606-28409-evidence-driven-llm-agent-for-c-to-synthesizable-c-conversion-and-verification.md create mode 100644 papers/items/2026-2606-28425-tool-use-enables-undetectable-steganography-in-multi-agent-llm-systems.md create mode 100644 papers/items/2026-2606-28430-building-to-the-test-coding-agents-deliver-what-you-check-not-what-you-requested.md create mode 100644 papers/items/2026-2606-28434-swe-mem-learning-adaptive-memory-management-for-long-horizon-coding-agents.md create mode 100644 papers/items/2026-2606-28436-dockerless-environment-free-program-verifier-for-coding-agents.md create mode 100644 papers/items/2026-2606-28450-llm-agents-security-duality-a-comprehensive-survey-of-self-security-and-empowere.md create mode 100644 papers/items/2026-2606-28456-is-lying-an-emergent-behaviour-in-llms-evidence-from-gaslighting-ai-agents-in-a-.md create mode 100644 papers/items/2026-2606-28467-an-agentic-ai-pipeline-for-appliance-level-energy-anomaly-detection-and-llm-driv.md create mode 100644 papers/items/2026-2606-28480-tua-bench-a-benchmark-for-general-purpose-terminal-use-agents.md create mode 100644 papers/items/2026-2606-28570-digitizing-coaching-intelligence-an-agentic-framework-for-holistic-athlete-profi.md create mode 100644 papers/items/2026-2606-28666-why-trust-your-agent-empirical-security-gains-from-trism-guided-agentic-workflow.md create mode 100644 papers/items/2026-2606-28679-capability-gates-are-not-authorization-confused-deputy-failures-in-llm-agent-fra.md create mode 100644 papers/items/2026-2606-28692-an-ai-agent-for-treatment-reasoning-over-a-biomedical-tool-universe.md create mode 100644 papers/items/2026-2606-28733-agentic-abstention-do-agents-know-when-to-stop-instead-of-act.md create mode 100644 papers/items/2026-2606-28739-agent-safety-is-action-alignment.md create mode 100644 papers/items/2026-2606-28781-hyphaedb-a-living-knowledge-topology-for-agent-first-memory.md create mode 100644 papers/items/2026-2606-28791-from-determinism-to-delegation-ai-native-software-engineering-and-the-evolution-.md create mode 100644 papers/items/2026-2606-28839-the-contagion-tensor-a-framework-for-measuring-output-distribution-coupling-in-m.md create mode 100644 papers/items/2026-2606-28841-lamp-lean-based-agentic-framework-with-mcp-and-proof-repair.md create mode 100644 papers/items/2026-2606-28896-a-task-driven-and-quality-assured-agent-framework-for-sar-data-generation.md create mode 100644 papers/items/2026-2606-28925-multi-agent-routing-as-set-valued-prediction-a-wildchat-benchmark-and-cost-aware.md create mode 100644 papers/items/2026-2606-28958-when-latent-agents-lie-kv-cache-integrity-in-multi-agent-llm-collaboration.md create mode 100644 papers/items/2026-2606-29014-customized-generative-ai-agent-for-transportation-engineering-practice-a-develop.md create mode 100644 papers/items/2026-2606-29026-preventing-error-propagation-in-multi-agent-ai-through-runtime-monitoring.md create mode 100644 papers/items/2026-2606-29030-memory-as-an-attack-surface-in-llm-agents-a-study-on-multiple-choice-question-an.md create mode 100644 papers/items/2026-2606-29116-characterizing-large-language-model-agentic-workflows-a-study-on-n8n-ecosystem.md create mode 100644 papers/items/2026-2606-29142-agent-security-meets-regulatory-reality-a-practitioner-systematization-of-autono.md create mode 100644 papers/items/2026-2606-29178-selective-memory-retention-for-long-horizon-llm-agents.md create mode 100644 papers/items/2026-2606-29193-a-multi-dataset-benchmark-for-evaluating-llm-agents-in-microservice-failure-diag.md create mode 100644 papers/items/2026-2606-29225-policyguard-a-dialogue-grounded-sub-agent-verifier-for-policy-adherence-in-llm-a.md create mode 100644 papers/items/2026-2606-29270-minority-sentinel-when-to-overturn-majority-voting-in-multi-agent-llm-debates.md create mode 100644 papers/items/2026-2606-29315-hierarchical-experimentalist-agents.md create mode 100644 papers/items/2026-2606-29354-when-llms-develop-languages-symbolic-communication-for-efficient-multi-agent-rea.md create mode 100644 papers/items/2026-2606-29445-bridging-videoqa-and-video-guided-agentic-tasks-via-generalized-keyframe-extract.md create mode 100644 papers/items/2026-2606-29495-cognitive-world-models-for-process-level-social-influence-evaluation.md create mode 100644 papers/items/2026-2606-29537-osworld2-0-benchmarking-computer-use-agents-on-long-horizon-real-world-tasks.md create mode 100644 papers/items/2026-2606-29654-budgeted-act-or-defer-multi-agent-llm-deliberation-with-local-reliability-bounds.md create mode 100644 papers/items/2026-2606-29719-a-diagnostic-framework-and-multi-evaluator-audit-of-evaluator-driven-preference-.md create mode 100644 papers/items/2026-2606-29742-microagent-context-augmented-multi-agent-framework-for-automatic-microservice-de.md create mode 100644 papers/items/2026-2606-29745-echo-learning-epistemically-adaptive-language-agents-with-turn-level-credit.md create mode 100644 papers/items/2026-2606-29746-deepmed-search-an-open-source-agentic-platform-for-medical-deep-research-with-in.md create mode 100644 papers/items/2026-2606-29762-do-recommendation-algorithms-work-when-users-are-llm-agents-a-case-study-on-molt.md create mode 100644 papers/items/2026-2606-29771-clqt-a-closed-loop-cost-aware-strategy-consistent-benchmark-for-diagnostic-evalu.md create mode 100644 papers/items/2026-2606-29774-analytic-concept-centric-memory-for-agentic-embodied-manipulation.md create mode 100644 papers/items/2026-2606-29778-mandol-an-agglomerative-agent-memory-system-for-long-term-conversations.md create mode 100644 papers/items/2026-2606-29788-memleak-diagnosing-information-leaks-in-multimodal-agent-memory.md create mode 100644 papers/items/2026-2606-29824-neural-procedural-memory-empowering-llm-agents-with-implicit-activation-steering.md create mode 100644 papers/items/2026-2606-29894-saber-math-automated-benchmark-for-information-retrieval-evaluation-in-mathemati.md create mode 100644 papers/items/2026-2606-29914-memdelta-controlled-baselines-and-hidden-confounds-in-agent-memory-evaluation.md create mode 100644 papers/items/2026-2606-29932-saga-scene-aware-goal-evolving-agents-for-long-horizon-civrealm-strategy-plannin.md create mode 100644 papers/items/2026-2606-29957-swe-together-evaluating-coding-agents-in-interactive-user-sessions.md create mode 100644 papers/items/2026-2606-29961-duomem-towards-capable-on-device-memory-agents-via-dual-space-distillation.md create mode 100644 papers/items/2026-2606-30005-llm-agents-are-latent-context-managers-eliciting-self-managed-context-via-a-prop.md create mode 100644 papers/items/2026-2606-30111-automating-the-design-of-embodied-agent-architectures.md create mode 100644 papers/items/2026-2606-30119-on-the-internet-nobody-knows-you-re-an-llm-bot-unmasking-web-agents-with-multi-l.md create mode 100644 papers/items/2026-2606-30185-dynamo-dynamic-skill-tool-evolution-for-vision-language-agents.md create mode 100644 papers/items/2026-2606-30251-taco-tool-augmented-credit-optimization-for-agentic-tool-use.md create mode 100644 papers/items/2026-2606-30259-multi-agentic-system-leveraging-open-source-llms-to-mitigate-disinformation-thre.md create mode 100644 papers/items/2026-2606-30266-towards-continual-motion-language-agents-lora-variants-for-incremental-motion-un.md create mode 100644 papers/items/2026-2606-30294-rehearsed-multi-agent-live-product-demonstrations-with-real-time-voice-question-.md create mode 100644 papers/items/2026-2606-30383-whose-side-is-your-agent-on-multi-party-principal-loyalty-in-llm-agents.md create mode 100644 papers/items/2026-2606-30454-collective-cooperation-without-individual-fidelity-in-llm-agents.md create mode 100644 papers/items/2026-2606-30524-the-illusion-of-agentic-complexity-in-readme-md-generation-evaluating-single-age.md create mode 100644 papers/items/2026-2606-30546-mas-lab-a-specification-driven-validation-framework-for-reliable-multi-agent-sys.md create mode 100644 papers/items/2026-2606-30555-linguistic-firewall-geometry-as-defense-in-multi-agent-systems-routing.md create mode 100644 papers/items/2026-2606-30560-tracelab-characterizing-coding-agent-workloads-for-llm-serving.md create mode 100644 papers/items/2026-2606-30566-forensic-trajectory-signatures-for-agent-memory-poisoning-detection.md create mode 100644 papers/items/2026-2606-30573-swe-interact-reimagining-swe-benchmarks-as-user-driven-long-horizon-coding-sessi.md create mode 100644 papers/items/2026-2606-30602-mesa-prioritizing-vulnerable-communication-channels-for-securing-multi-agent-sys.md create mode 100644 papers/items/2026-2606-30616-scaling-the-horizon-not-the-parameters-reaching-trillion-parameter-performance-w.md create mode 100644 papers/items/2026-2606-30639-self-evolving-world-models-for-llm-agent-planning.md create mode 100644 papers/items/2026-2606-30697-lumos-a-semantic-operating-system-layer-for-accessibility-grounded-ai-agents.md create mode 100644 papers/items/2026-2606-30755-understanding-and-evaluating-claw-like-agent-security-through-a-computer-systems.md create mode 100644 papers/items/2026-2606-30840-contrastive-reflection-for-iterative-prompt-optimization.md create mode 100644 papers/items/2026-2606-30877-a-systematic-approach-to-multi-agent-ai-from-advanced-regulatory-control-theory-.md create mode 100644 papers/items/2026-2606-30887-training-therapeutic-judges-and-multi-agent-systems-for-human-aligned-mental-hea.md create mode 100644 papers/items/2026-2606-30906-investigating-multi-agent-deliberation-in-law.md create mode 100644 papers/items/2026-2606-30949-agrefactor-self-evolving-agentic-workflow-for-hls-compatibility-and-performance.md create mode 100644 papers/items/2026-2606-30970-behavioral-governance-for-autonomous-ai-agents-the-agentbound-framework.md create mode 100644 papers/items/2026-2606-30986-the-organizational-behavior-of-agentic-ai-collective-intelligence-in-human-agent.md create mode 100644 papers/items/2026-2606-31046-openlife-toward-open-world-artificial-life-with-autonomous-llm-agents.md create mode 100644 papers/items/2026-2606-31073-multiuav-plat-an-llm-oriented-platform-benchmark-and-framework-for-multi-uav-col.md create mode 100644 papers/items/2026-2606-31085-ddiagents-mechanism-conditioned-context-flow-for-drug-drug-interaction-predictio.md create mode 100644 papers/items/2026-2606-31134-beyond-the-library-an-agentic-framework-for-autoformalizing-research-mathematics.md create mode 100644 papers/items/2026-2606-31174-clawarena-team-benchmarking-subagent-orchestration-and-dynamic-workflows-in-lang.md create mode 100644 papers/items/2026-2606-31179-healthagentbench-a-unified-benchmark-suite-of-realistic-agentic-healthcare-envir.md create mode 100644 papers/items/2026-2606-31200-agentic-rag-vlm-affordance-aware-retrieval-augmented-generation-with-self-reflec.md create mode 100644 papers/items/2026-2606-31209-long-term-traffic-simulation-via-structured-autoregressive-modeling.md create mode 100644 papers/items/2026-2606-31227-securing-the-ai-agent-a-unified-framework-for-multi-layer-agent-red-teaming.md create mode 100644 papers/items/2026-2606-31229-agentic-ideation-sample-efficient-agentic-trajectories-synthesis-for-scientific-.md create mode 100644 papers/items/2026-2606-31252-embodied-cad-solver-grounded-llm-agents-for-parametric-b-rep-assembly-modeling.md create mode 100644 papers/items/2026-2606-31314-a-novel-method-for-differential-algebraic-dynamic-model-discovery-in-power-syste.md create mode 100644 papers/items/2026-2606-31339-verification-gated-agentic-mission-state-governance-for-intelligent-industrial-m.md create mode 100644 papers/items/2026-2606-31410-xiaomi-gui-0-technical-report.md create mode 100644 papers/items/2026-2606-31471-think-while-you-map-asynchronous-vision-language-agents-for-incremental-3d-scene.md create mode 100644 papers/items/2026-2606-31612-what-memory-do-gui-agents-really-need-from-passive-records-to-active-task-drivin.md create mode 100644 papers/items/2026-2606-31635-a-tutorial-on-autonomous-fault-tolerant-control-using-knowledge-grounded-llm-age.md create mode 100644 papers/items/2026-2606-31639-a-lifecycle-and-application-stack-survey-of-large-language-model-vulnerabilities.md create mode 100644 papers/items/2026-2606-31648-think-in-english-answer-in-korean-efficient-adaptation-of-multilingual-tool-usin.md create mode 100644 papers/items/2026-2606-31650-echo-prune-to-act-trace-to-learn-with-selective-turn-memory-in-agentic-rl.md create mode 100644 papers/items/2026-2606-31665-forecastagentsearch-towards-a-multi-expert-agent-search-system-for-geopolitical-.md create mode 100644 papers/items/2026-2606-31693-shopx-a-foundation-model-for-intent-to-item-fulfillment-in-agentic-shopping.md create mode 100644 papers/items/2026-2606-31744-a-conversational-agentic-interface-to-physics-based-household-digital-twins-for-.md create mode 100644 papers/items/2026-2606-31767-jeto-bench-a-reproducible-benchmark-for-execution-time-improvement-patches-in-ja.md create mode 100644 papers/items/2026-2606-31831-an-agentic-ai-framework-to-accelerate-scientific-discovery-in-plant-phenotyping.md create mode 100644 papers/items/2026-2606-31916-theory-of-mind-and-persuasion-beyond-conversation-assessing-the-capacity-of-llms.md create mode 100644 papers/items/2026-2606-31980-digitalcoach-communication-and-grounding-gaps-in-human-and-agentic-computer-use-.md create mode 100644 papers/items/2026-2606-32025-generative-skill-composition-for-llm-agents.md create mode 100644 papers/items/2026-2606-32034-qval-cheaply-evaluating-dense-supervision-signals-for-long-horizon-llm-agents.md create mode 100644 papers/items/2026-2607-00038-stop-hand-holding-your-coding-agent-engineering-the-loops-that-replace-step-by-s.md create mode 100644 papers/items/2026-2607-00041-atm-cid-brokered-pre-write-admission-for-multi-agent-code-co-synthesis.md create mode 100644 papers/items/2026-2607-00233-from-signals-to-structure-how-memory-architecture-drives-language-emergence-in-l.md create mode 100644 papers/items/2026-2607-00255-slm-llm-or-agentic-ai-toward-intelligent-uav-enabled-wpt-systems-in-low-altitude.md create mode 100644 papers/items/2026-2607-00297-epc-a-standardized-protocol-for-measuring-evaluator-preference-dynamics-in-llm-a.md create mode 100644 papers/items/2026-2607-00334-managed-autonomy-at-runtime-gear-based-safety-and-governance-for-single-and-mult.md create mode 100644 papers/items/2026-2607-00345-registry-governed-agent-lifecycle-completing-eddops-with-evaluation-drivenregist.md create mode 100644 papers/items/2026-2607-00407-personalization-as-inverse-planning-learning-latent-design-intents-for-agentic-s.md create mode 100644 papers/items/2026-2607-00422-kidnaprag-a-black-box-attack-for-hijacking-reasoning-in-agentic-retrieval-augmen.md create mode 100644 papers/items/2026-2607-00436-phreeqc-mcq-200-a-diagnostic-benchmark-for-tool-augmented-scientific-simulator-a.md create mode 100644 papers/items/2026-2607-00440-minos-a-multi-agent-collaborative-framework-for-provenance-based-backward-tracki.md create mode 100644 papers/items/2026-2607-00454-agri-sage-simulation-grounded-multi-agent-llm-for-context-aware-agricultural-adv.md create mode 100644 papers/items/2026-2607-00502-a-task-state-representation-for-long-horizon-mobile-gui-agents.md create mode 100644 papers/items/2026-2607-00555-rise-from-the-ashes-llm-based-static-analysis-for-deep-learning-framework-bugs.md create mode 100644 papers/items/2026-2607-00604-vehicle-routing-problem-meets-large-language-models-an-overview-and-perspectives.md create mode 100644 papers/items/2026-2607-00627-agi-maze-as-a-benchmark-framework-for-world-modeling-agents.md create mode 100644 papers/items/2026-2607-00692-self-gc-self-governing-context-for-long-horizon-llm-agents.md create mode 100644 papers/items/2026-2607-00911-from-registry-to-repository-how-ai-agent-skills-are-written-adapted-and-maintain.md create mode 100644 papers/items/2026-2607-00918-from-personas-to-plot-character-grounded-multi-agent-story-generation-for-long-f.md create mode 100644 papers/items/2026-2607-00939-leveraging-llm-based-agentic-systems-to-generate-quantum-applications-for-test-o.md create mode 100644 papers/items/2026-2607-00972-bayesian-uncertainty-propagation-for-agentic-rag-pipelines-a-proof-of-concept-st.md create mode 100644 papers/items/2026-2607-00990-swe-doctor-guiding-software-engineering-agents-with-runtime-diagnosis-from-multi.md create mode 100644 papers/items/2026-2607-01047-conversable-complexity-agentic-llm-collectives-as-interpretable-substrates.md create mode 100644 papers/items/2026-2607-01061-agentic-generation-of-verifiable-rules-for-deterministic-self-expanding-reaction.md create mode 100644 papers/items/2026-2607-01071-memsyco-bench-benchmarking-sycophancy-in-agent-memory.md create mode 100644 papers/items/2026-2607-01084-can-agents-generalize-to-the-open-world-unveiling-the-fragility-of-static-traini.md create mode 100644 papers/items/2026-2607-01120-next-generation-agentic-reinforcement-learning-systems-enable-self-evolving-agen.md create mode 100644 papers/items/2026-2607-01211-are-performance-optimization-benchmarks-reliably-measuring-coding-agents.md create mode 100644 papers/items/2026-2607-01213-reporescue-an-empirical-study-of-llm-agents-on-whole-repository-compatibility-re.md create mode 100644 papers/items/2026-2607-01366-auto-fl-research-agentic-search-for-federated-learning-algorithms.md create mode 100644 papers/items/2026-2607-01421-risk-architecture-for-ai-native-engineering-teams-an-organizational-framework-fo.md create mode 100644 papers/items/2026-2607-01510-janus-a-playground-for-user-involved-agentic-permission-management.md create mode 100644 papers/items/2026-2607-01523-multi-head-recurrent-memory-agents.md create mode 100644 papers/items/2026-2607-01531-opine-world-programmatic-world-modeling-with-ontology-error-prioritized-interact.md create mode 100644 papers/items/2026-2607-01600-boundary-sync-measuring-communication-induced-representational-coupling-in-multi.md create mode 100644 papers/items/2026-2607-01640-agentflow-building-agent-dependency-graphs-for-static-analysis-of-agent-programs.md create mode 100644 papers/items/2026-2607-01641-when-agents-do-not-stop-uncovering-infinite-agentic-loops-in-llm-agents.md create mode 100644 papers/items/2026-2607-01661-diverse-evidence-better-forecasts-multi-agent-deliberation-under-information-asy.md create mode 100644 papers/items/2026-2607-01668-verichat-an-agentic-conversational-ai-assistant-for-hardware-security-verificati.md create mode 100644 papers/items/2026-2607-01709-comfyclaw-self-evolving-skill-harnesses-for-image-generation-workflows.md create mode 100644 papers/items/2026-2607-01766-simworlds-a-multi-agent-system-for-dynamic-3d-scene-creation.md create mode 100644 papers/items/2026-2607-01767-repair-the-amplifier-not-the-symptom-stable-world-model-correction-for-agent-rol.md create mode 100644 papers/items/2026-2607-01788-krca-an-efficient-root-cause-analysis-system-in-hyper-scale-microservice-systems.md create mode 100644 papers/items/2026-2607-01793-safety-testing-llm-agents-at-scale-from-risk-discovery-to-evidence-grounded-veri.md create mode 100644 papers/items/2026-2607-01812-to-master-an-llm-agent-framework-for-automated-topology-optimization.md create mode 100644 papers/items/2026-2607-01874-skillcoach-self-evolving-rubrics-for-evaluating-and-enhancing-agentic-skill-use.md create mode 100644 papers/items/2026-2607-01916-contextsniper-anttrail-s-token-efficient-code-memory-for-repository-level-progra.md create mode 100644 papers/items/2026-2607-01929-beyond-textual-repository-exploration-dual-modal-structural-reasoning-for-agenti.md create mode 100644 papers/items/2026-2607-01935-a-tma-decoupling-state-aware-memory-failures-in-long-term-agent-memory.md create mode 100644 papers/items/2026-2607-02032-pace-a-proxy-for-agentic-capability-evaluation.md create mode 100644 papers/items/2026-2607-02186-ua-chatdev-uncertainty-aware-multi-agent-collaboration-for-reliable-software-dev.md create mode 100644 papers/items/2026-2607-02210-criticality-based-guard-rail-validation-for-ai-agent-decisions-in-autonomous-tel.md create mode 100644 papers/items/2026-2607-02245-copewell-a-multi-agent-swarm-architecture-for-equitable-mental-wellness-support.md create mode 100644 papers/items/2026-2607-02255-agenticsts-a-bounded-memory-testbed-for-long-horizon-llm-agents.md create mode 100644 papers/items/2026-2607-02294-coding-agents-are-guessing-measuring-action-boundary-violations-in-underspecifie.md create mode 100644 papers/items/2026-2607-02381-hulat2-at-mer-trans-2026-governed-multi-agent-simplification-for-spanish-easy-to.md create mode 100644 papers/items/2026-2607-02448-agentscad-automated-design-for-manufacturing-of-fdm-parts-via-multi-agent-llm-re.md create mode 100644 papers/items/2026-2607-02453-adoption-and-ecosystem-health-a-longitudinal-analysis-of-open-source-multi-agent.md create mode 100644 papers/items/2026-2607-02507-what-llm-agents-say-when-no-one-is-watching-social-structure-and-latent-objectiv.md create mode 100644 papers/items/2026-2607-02577-benchmarking-the-benchmarks-a-validity-audit-of-tool-calling-evaluation.md create mode 100644 papers/items/2026-2607-02579-when-not-to-write-memory-governing-false-promotion-from-correlated-agent-traces.md create mode 100644 papers/items/2026-2607-02599-agentltl-a-trace-verification-framework-for-measuring-enforcing-and-training-pro.md create mode 100644 papers/items/2026-2607-02606-chainswe-benchmarking-coding-agents-on-multi-bug-software-maintenance.md create mode 100644 papers/items/2026-2607-02684-can-coding-agents-implement-missed-compiler-optimizations-evaluating-llm-agents-.md create mode 100644 papers/items/2026-2607-02689-s-ember-a-large-scale-benchmark-for-streaming-egocentric-memory-retrieval.md create mode 100644 papers/items/2026-2607-02703-llmoxie-exploring-agentic-ai-for-scientific-software-development.md create mode 100644 papers/items/2026-2607-02716-evaluating-large-language-models-for-decision-making-in-agent-based-urban-mobili.md create mode 100644 papers/items/2026-2607-02807-swarmresearch-orchestrating-coding-agents-for-open-ended-discovery.md create mode 100644 papers/items/2026-2607-02846-object-centric-environment-modeling-for-agentic-tasks.md create mode 100644 papers/items/2026-2607-02857-mosaic-knowledge-guided-cli-command-composition-attack-in-llm-coding-agents.md create mode 100644 papers/items/2026-2607-02879-medcalc-pro-solving-complex-medical-calculations-with-llm-agents.md create mode 100644 papers/items/2026-2607-02882-diagnosis-driven-automatic-repair-for-agentic-workflow-via-symbolic-inference.md create mode 100644 papers/items/2026-2607-02911-coact-action-preserving-observation-compression-for-coding-agents.md create mode 100644 papers/items/2026-2607-02927-videosearcher-empowering-video-deep-research-with-multi-tool-agentic-reasoning-v.md create mode 100644 papers/items/2026-2607-02942-a-workflow-aware-serving-layer-for-agentic-applications.md create mode 100644 papers/items/2026-2607-03105-orbit-q-dual-axis-benchmarking-of-autonomous-agents-in-scientific-quantum-progra.md create mode 100644 papers/items/2026-2607-03162-apeb-benchmarking-personalization-ability-of-large-language-model-agents.md create mode 100644 papers/items/2026-2607-03220-contra-red-teaming-configurations-of-personalizable-agents.md create mode 100644 papers/items/2026-2607-03233-agentic-and-generative-ai-for-open-source-intelligence-and-cyber-investigations-.md create mode 100644 papers/items/2026-2607-03269-agentic-secpbft-agentic-ai-driven-proactive-security-framework-for-wireless-pbft.md create mode 100644 papers/items/2026-2607-03316-is-agentic-code-review-helpful-mining-developers-feedback-to-coderabbit-reviews-.md create mode 100644 papers/items/2026-2607-03333-spork-self-speculative-forking-to-accelerate-agentic-llm-inference.md create mode 100644 papers/items/2026-2607-03423-securing-multi-tool-ai-agent-chains-with-dynamic-real-time-compositional-policie.md create mode 100644 papers/items/2026-2607-03441-no-time-like-the-present-agentic-test-time-training-for-llm-agents.md create mode 100644 papers/items/2026-2607-03510-cage-1-control-assurance-and-governance-evaluation-for-enterprise-agentic-ai.md create mode 100644 papers/items/2026-2607-03525-gameenginebench-evaluating-coding-agents-on-real-c-runtime-environments.md create mode 100644 papers/items/2026-2607-03601-archeval-measuring-ai-agents-as-computer-architects.md create mode 100644 papers/items/2026-2607-03628-swarm-driven-multi-agent-reasoning-for-smart-city-security.md create mode 100644 papers/items/2026-2607-03691-don-t-blame-the-large-language-model-how-scaffolding-evolution-shapes-coding-age.md create mode 100644 papers/items/2026-2607-03695-social-networks-of-llm-agents.md create mode 100644 papers/items/2026-2607-03702-agent-reinforcement-learning-via-pivotal-aware-self-feedback-retry.md create mode 100644 papers/items/2026-2607-03726-selfmem-self-optimizing-memory-for-ai-agents.md create mode 100644 papers/items/2026-2607-03821-dualview-preventing-indirect-prompt-injection-in-personal-ai-agents.md create mode 100644 papers/items/2026-2607-03853-cograd-a-cognitively-inspired-multi-agent-framework-for-radiology-report-generat.md create mode 100644 papers/items/2026-2607-03953-the-remarkable-effectiveness-of-providing-ai-agents-with-natural-language-tools-.md create mode 100644 papers/items/2026-2607-03968-refused-in-chat-written-in-code-workflow-level-jailbreak-construction-in-ide-cod.md create mode 100644 papers/items/2026-2607-04009-physminer-an-agentic-ai-framework-for-discovering-turbulence-physics.md create mode 100644 papers/items/2026-2607-04034-the-i-don-t-know-filter-enhancing-agentic-reliability-in-function-calling.md create mode 100644 papers/items/2026-2607-04089-placemem-toward-a-compute-aware-memory-plane-for-lifelong-agents.md create mode 100644 papers/items/2026-2607-04149-beyond-scene-priors-fine-grained-traffic-scene-reasoning-with-benchmarking-and-q.md create mode 100644 papers/items/2026-2607-04162-ace-agentic-control-for-embodied-manipulation-via-zero-shot-workflow-reasoning.md create mode 100644 papers/items/2026-2607-04212-an-evaluation-of-role-based-multi-agent-code-generation-on-repository-scale-prob.md create mode 100644 papers/items/2026-2607-04219-agentic-iot-architectures-applications-and-challenges-toward-the-internet-of-age.md create mode 100644 papers/items/2026-2607-04240-biological-motifs-for-agentic-control.md create mode 100644 papers/items/2026-2607-04293-causalgame-benchmarking-causal-thinking-of-llm-agents-in-games.md create mode 100644 papers/items/2026-2607-04334-do-gui-agents-believe-their-eyes-diagnosing-state-belief-reliance-on-pixels-vers.md create mode 100644 papers/items/2026-2607-04391-memory-orchestrated-semantic-system-moss-an-auditable-agentic-memory-architectur.md create mode 100644 papers/items/2026-2607-04394-mechmath-agent-team-llm-driven-agents-for-mathematical-research.md create mode 100644 papers/items/2026-2607-04395-nki-agent-domain-specific-fine-tuning-and-agentic-tool-use-for-neuron-kernel-gen.md create mode 100644 papers/items/2026-2607-04426-ace-brain-0-5-a-unified-embodied-foundational-model-for-physical-agentic-ai.md create mode 100644 papers/items/2026-2607-04433-autonomous-information-seeking-a-roadmap-for-agentic-recommender-systems.md create mode 100644 papers/items/2026-2607-04470-regime-conditional-stabilisation-of-llm-augmented-cooperative-multi-agent-reinfo.md create mode 100644 papers/items/2026-2607-04528-measuring-harness-induced-belief-divergence-in-multi-step-llm-agents.md create mode 100644 papers/items/2026-2607-04569-llms-for-agentic-home-energy-management.md create mode 100644 papers/items/2026-2607-04617-mrms-a-multi-resolution-memory-substrate-for-long-lived-ai-agents.md create mode 100644 papers/items/2026-2607-04623-can-llms-really-recover-microservice-failures-a-recovery-aware-evaluation-of-dia.md create mode 100644 papers/items/2026-2607-04686-toolfailbench-diagnosing-tool-use-failures-in-llm-agents.md create mode 100644 papers/items/2026-2607-04697-ai-agent-pull-requests-on-github-frequency-structure-and-merge-conflict-rates.md create mode 100644 papers/items/2026-2607-04713-rspo-reward-swap-policy-optimization-for-multi-turn-llm-agents.md create mode 100644 papers/items/2026-2607-04963-stapo-selective-trajectory-aware-policy-optimization-for-llm-agent-training.md create mode 100644 papers/items/2026-2607-05001-tactic-kg-toward-small-agent-teams-for-cyber-threat-intelligence-knowledge-graph.md create mode 100644 papers/items/2026-2607-05029-your-agent-s-memories-are-not-its-own-forged-reasoning-attacks-on-llm-agent-memo.md create mode 100644 papers/items/2026-2607-05055-toward-trustworthy-large-language-model-agents-in-healthcare.md create mode 100644 papers/items/2026-2607-05120-agent-data-injection-attacks-are-realistic-threats-to-ai-agents.md create mode 100644 papers/items/2026-2607-05132-when-agents-lie-premeditation-persistence-and-exploitation-in-repeated-games.md create mode 100644 papers/items/2026-2607-05174-agentgym2-benchmarking-large-language-model-agents-in-de-idealized-real-world-en.md create mode 100644 papers/items/2026-2607-05188-latent-programming-horizons-in-coding-agents.md create mode 100644 papers/items/2026-2607-05202-evoagentbench-benchmarking-agent-self-evolution-via-ability-transfer.md create mode 100644 papers/items/2026-2607-05297-metaskill-evolve-recursive-self-improvement-of-llm-agents-via-two-timescale-meta.md create mode 100644 papers/items/2026-2607-05318-pisas-benchmarking-contextual-integrity-in-multi-user-agentic-systems.md create mode 100644 papers/items/2026-2607-05363-sovereignpa-bench-evaluating-user-owned-personal-agents-under-evolving-intent-pl.md create mode 100644 papers/items/2026-2607-05378-compactionrl-reinforcement-learning-with-context-compaction-for-long-horizon-age.md create mode 100644 papers/items/2026-2607-05391-llm-as-a-verifier-a-general-purpose-verification-framework.md create mode 100644 papers/items/2026-2607-05428-charlie-an-on-premise-multi-agent-retrieval-augmented-generation-system-for-evid.md create mode 100644 papers/items/2026-2607-05456-prompt-to-paper-agentic-ai-system-for-bioinformatics.md create mode 100644 papers/items/2026-2607-05458-learning-to-control-llm-agent-harnesses-with-offline-reinforcement-learning.md create mode 100644 papers/items/2026-2607-05518-aiauthz-off-host-identity-bound-authorization-for-ai-agents.md create mode 100644 papers/items/2026-2607-05659-agents-with-feelings-personality-and-emotion-in-multi-agent-software-teams.md create mode 100644 papers/items/2026-2607-05666-what-do-ai-agents-actually-change-an-empirical-taxonomy-of-mutation-patterns-in-.md create mode 100644 papers/items/2026-2607-05677-from-conversation-to-contribution-characterizing-coding-agent-in-open-source-sof.md create mode 100644 papers/items/2026-2607-05690-memory-in-the-loop-in-process-retrieval-as-extendedworking-memory-for-language-a.md create mode 100644 papers/items/2026-2607-05743-the-balkanization-of-execution-security-research-for-ai-coding-agents-isolation-.md create mode 100644 papers/items/2026-2607-05772-detecting-vulnerability-inducing-commits-via-multi-stage-reasoning-with-llm-base.md create mode 100644 papers/items/2026-2607-05773-beyond-static-evaluation-building-simulation-environments-for-scalable-agentic-r.md create mode 100644 papers/items/2026-2607-05775-beyond-the-leaderboard-a-synthesis-of-tool-use-planning-and-reasoning-failures-i.md create mode 100644 papers/items/2026-2607-05794-from-passive-retrieval-to-active-memory-navigation-learning-to-use-memory-as-a-s.md create mode 100644 papers/items/2026-2607-05805-onnes-a-physics-grounded-multi-agent-llm-simulator-for-cryogenic-fault-diagnosis.md create mode 100644 papers/items/2026-2607-05915-pcbworld-a-benchmark-environment-for-engine-grounded-pcb-design-automation.md create mode 100644 papers/items/2026-2607-06000-context-to-execution-integrity-for-llm-agents.md create mode 100644 papers/items/2026-2607-06001-information-limits-and-attractor-dynamics-in-economies-of-frontier-llm-agents-a-.md create mode 100644 papers/items/2026-2607-06008-polyworkbench-benchmarking-multilingual-long-horizon-llm-agents.md create mode 100644 papers/items/2026-2607-06080-from-blueprint-to-reality-modeling-and-applying-putnam-s-social-capital-theory-w.md create mode 100644 papers/items/2026-2607-06101-agents-that-teach-towards-designing-incidental-learning-back-into-ai-assisted-so.md create mode 100644 papers/items/2026-2607-06118-webretriever-a-large-scale-comprehensive-benchmark-for-efficient-web-agent-evalu.md create mode 100644 papers/items/2026-2607-06140-curateevo-data-curation-evolving-for-agentic-post-training.md create mode 100644 papers/items/2026-2607-06157-llm-agents-for-deliberative-collaboration-a-study-on-joint-decision-making-under.md create mode 100644 papers/items/2026-2607-06195-logichunter-testing-llm-agent-frameworks-with-an-agentic-oracle.md create mode 100644 papers/items/2026-2607-06223-information-gain-based-rollout-policy-optimization-an-adaptive-tree-structured-r.md create mode 100644 papers/items/2026-2607-06273-agenttether-graph-guided-diagnosis-and-runtime-intervention-for-reliable-llm-age.md create mode 100644 papers/items/2026-2607-06341-harnessing-code-agents-for-automatic-software-verification.md create mode 100644 papers/items/2026-2607-06411-rubench-a-repository-level-agentic-coding-benchmark-with-natively-authored-russi.md create mode 100644 papers/items/2026-2607-06413-an-experimental-design-approach-to-evaluating-agentic-ai-s-autonomous-model-disc.md create mode 100644 papers/items/2026-2607-06452-from-voting-to-agent-collaboration-answer-type-aware-llm-pipelines-for-bioasq-14.md create mode 100644 tools/collection/collect_arxiv.py create mode 100644 tools/collection/promote_arxiv_manifest.py diff --git a/collection-runs/2026-07-08-arxiv-expanded-sweep.md b/collection-runs/2026-07-08-arxiv-expanded-sweep.md new file mode 100644 index 0000000..e21f123 --- /dev/null +++ b/collection-runs/2026-07-08-arxiv-expanded-sweep.md @@ -0,0 +1,55 @@ +# Collection Run: 2026-07-08 arXiv Expanded Sweep + +status: completed + +## Scope + +- source: arXiv API +- date window: 2025-07-08 to 2026-07-08 +- query groups: 15 +- unique candidates seen: 1506 +- high-relevance candidates promoted to paper items: 970 +- remaining reserve candidates: 536 +- new paper files written in this run: 968 + +## Query Groups + +- llm-agent +- language-agent +- ai-agent +- agentic-ai +- agent-evaluation +- agent-memory +- tool-use +- function-calling +- coding-agent +- web-gui-agent +- multi-agent-llm +- agent-safety +- rag-agent +- planning-agent +- autonomous-agent-llm + +## Outputs + +- `data/arxiv-agent-candidates-2025-07-08-to-2026-07-08.json` +- `data/arxiv-agent-papers-2025-07-08-to-2026-07-08.json` +- `papers/corpus-summary-2026-07-08.md` +- 968 generated `papers/items/*.md` files + +## Method Notes + +- The candidate manifest stores title, arXiv URL, authors, categories, dates, matched query groups, auto topics, score, and relevance. +- It intentionally does not store abstracts, to avoid turning the repo into a raw abstract mirror. +- Generated paper items are `queued`; they are collection records, not reviewed summaries. +- Auto tags are recall-oriented and need later cleanup. +- The first API pass selected the top 400; after inspecting score distribution, all 970 `relevance=high` candidates were promoted locally from the candidate manifest. +- arXiv returned 429 during a later refresh attempt, so the collector now includes retry/backoff controls and the promotion step can run offline. + +## First Trend Takeaways + +- Evaluation and benchmarking dominate the corpus. +- Memory is now a central systems topic, including governance, poisoning, provenance, and procedural memory. +- Agent safety has moved toward runtime risks: tool leakage, prompt injection, data injection, memory poisoning, and action-boundary violations. +- Coding-agent evaluation is moving beyond single-issue SWE-Bench tasks. +- GUI/computer-use agents and workflow/enterprise agents are now large enough to deserve separate tracks. diff --git a/data/arxiv-agent-candidates-2025-07-08-to-2026-07-08.json b/data/arxiv-agent-candidates-2025-07-08-to-2026-07-08.json new file mode 100644 index 0000000..dc99bd0 --- /dev/null +++ b/data/arxiv-agent-candidates-2025-07-08-to-2026-07-08.json @@ -0,0 +1,53788 @@ +[ + { + "id": "2606.08340", + "title": "Benchmarking Open-Ended Multi-Agent Coordination in Language Agents", + "url": "https://arxiv.org/abs/2606.08340", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Kale-ab Abebe Tessera", + "Andras Szecsenyi", + "Cameron Barker", + "Alexander Rutherford", + "Davide Paglieri", + "Aidan Scannell", + "Henry Gouk", + "Elliot J. Crowley", + "Tim Rocktäschel", + "Amos Storkey" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 28, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "language-agent", + "planning-agent" + ], + "arxiv_id": "2606.08340", + "source": "arxiv", + "source_id": "arxiv:2606.08340", + "pdf_url": "https://arxiv.org/pdf/2606.08340", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.20833", + "title": "MemGym: a Long-Horizon Memory Environment for LLM Agents", + "url": "https://arxiv.org/abs/2605.20833", + "published": "2026-05-20", + "updated": "2026-05-20", + "authors": [ + "Wujiang Xu", + "Yu Wang", + "Kai Mei", + "Kaiqu Liang", + "Zhenting Wang", + "Mingyu Jin", + "Han Zhang", + "Shi-Xiong Zhang", + "Wenyue Hua", + "Sambit Sahu", + "Dimitris N. Metaxas" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 26, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.20833", + "source": "arxiv", + "source_id": "arxiv:2605.20833", + "pdf_url": "https://arxiv.org/pdf/2605.20833", + "primary_query": "agent-memory" + }, + { + "id": "2607.06008", + "title": "PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2607.06008", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Hongliang Li", + "Yijin Liu", + "Zhiwei Zhang", + "Zihe Liu", + "Xinyue Lou", + "Jinan Xu", + "Fandong Meng", + "Kaiyu Huang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 25, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2607.06008", + "source": "arxiv", + "source_id": "arxiv:2607.06008", + "pdf_url": "https://arxiv.org/pdf/2607.06008", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.28425", + "title": "Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems", + "url": "https://arxiv.org/abs/2606.28425", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Jimmy Laurence Rippin", + "Simon C. Marshall", + "David Demitri Africa", + "Christian Schroeder de Witt" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use" + ], + "score": 25, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "autonomous-agent-llm", + "multi-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.28425", + "source": "arxiv", + "source_id": "arxiv:2606.28425", + "pdf_url": "https://arxiv.org/pdf/2606.28425", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24937", + "title": "The Hitchhiker's Guide to Agentic AI: From Foundations to Systems", + "url": "https://arxiv.org/abs/2606.24937", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Haggai Roitman" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.IR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 25, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm", + "rag-agent", + "tool-use" + ], + "arxiv_id": "2606.24937", + "source": "arxiv", + "source_id": "arxiv:2606.24937", + "pdf_url": "https://arxiv.org/pdf/2606.24937", + "primary_query": "agentic-ai" + }, + { + "id": "2606.10749", + "title": "Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation", + "url": "https://arxiv.org/abs/2606.10749", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yuchen Ling", + "Shengcheng Yu", + "Zhenyu Chen", + "Chunrong Fang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 25, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "planning-agent" + ], + "arxiv_id": "2606.10749", + "source": "arxiv", + "source_id": "arxiv:2606.10749", + "pdf_url": "https://arxiv.org/pdf/2606.10749", + "primary_query": "agent-safety" + }, + { + "id": "2606.29824", + "title": "Neural Procedural Memory: Empowering LLM Agents with Implicit Activation Steering", + "url": "https://arxiv.org/abs/2606.29824", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Chengfeng Zhao", + "Yuqiao Tan", + "Shizhu He", + "Yequan Wang", + "Jun Zhao", + "Kang Liu" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "agent-memory", + "autonomous-agent-llm", + "llm-agent", + "rag-agent" + ], + "arxiv_id": "2606.29824", + "source": "arxiv", + "source_id": "arxiv:2606.29824", + "pdf_url": "https://arxiv.org/pdf/2606.29824", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.28011", + "title": "From Detection to Action: Using LLM Agents for Fault-Tolerant Control", + "url": "https://arxiv.org/abs/2606.28011", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Javal Vyas", + "Milapji Singh Gill", + "Artan Markaj", + "Felix Gehlhoff", + "Mehmet Mercangöz" + ], + "categories": [ + "eess.SY", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent", + "multi-agent-llm", + "planning-agent", + "rag-agent" + ], + "arxiv_id": "2606.28011", + "source": "arxiv", + "source_id": "arxiv:2606.28011", + "pdf_url": "https://arxiv.org/pdf/2606.28011", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20401", + "title": "PowerAgentBench-Dyn: A Benchmark for Agentic AI in Power System Dynamic Studies", + "url": "https://arxiv.org/abs/2606.20401", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Qian Zhang", + "Andrea Pomarico", + "Costas Mylonas", + "Magda Foti", + "Alberto Berizzi", + "Le Xie" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.20401", + "source": "arxiv", + "source_id": "arxiv:2606.20401", + "pdf_url": "https://arxiv.org/pdf/2606.20401", + "primary_query": "agentic-ai" + }, + { + "id": "2606.18789", + "title": "PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies", + "url": "https://arxiv.org/abs/2606.18789", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Costas Mylonas", + "Magda Foti", + "Andrea Pomarico", + "Matheus Duarte", + "Qian Zhang", + "Emmanouel Varvarigos" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2606.18789", + "source": "arxiv", + "source_id": "arxiv:2606.18789", + "pdf_url": "https://arxiv.org/pdf/2606.18789", + "primary_query": "agentic-ai" + }, + { + "id": "2606.08274", + "title": "Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents", + "url": "https://arxiv.org/abs/2606.08274", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Safia Baloch", + "Rahemeen Khan" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2606.08274", + "source": "arxiv", + "source_id": "arxiv:2606.08274", + "pdf_url": "https://arxiv.org/pdf/2606.08274", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.05775", + "title": "Beyond the Leaderboard: A Synthesis of Tool-Use, Planning, and Reasoning Failures in Large Language Model Agents", + "url": "https://arxiv.org/abs/2607.05775", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Wael Albayaydh", + "Rui Zhao", + "Ivan Flechais" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "embodied-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2607.05775", + "source": "arxiv", + "source_id": "arxiv:2607.05775", + "pdf_url": "https://arxiv.org/pdf/2607.05775", + "primary_query": "llm-agent" + }, + { + "id": "2607.02255", + "title": "AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2607.02255", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Xiangchen Cheng", + "Yunwei Jiang", + "Jianwen Sun", + "Zizhen Li", + "Chuanhao Li", + "Xiangcheng Cao", + "Yihao Liu", + "Fanrui Zhang", + "Li Jin", + "Kaipeng Zhang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.02255", + "source": "arxiv", + "source_id": "arxiv:2607.02255", + "pdf_url": "https://arxiv.org/pdf/2607.02255", + "primary_query": "llm-agent" + }, + { + "id": "2606.28791", + "title": "From Determinism to Delegation: AI-Native Software Engineering and the Evolution of the Agentic Engineer", + "url": "https://arxiv.org/abs/2606.28791", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Mamdouh Alenezi" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "autonomous-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.28791", + "source": "arxiv", + "source_id": "arxiv:2606.28791", + "pdf_url": "https://arxiv.org/pdf/2606.28791", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26614", + "title": "HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization", + "url": "https://arxiv.org/abs/2606.26614", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Kuangshi Ai", + "Patrick Phuoc Do", + "Chaoli Wang" + ], + "categories": [ + "cs.HC", + "cs.AI", + "cs.GR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "planning-agent" + ], + "arxiv_id": "2606.26614", + "source": "arxiv", + "source_id": "arxiv:2606.26614", + "pdf_url": "https://arxiv.org/pdf/2606.26614", + "primary_query": "multi-agent-llm" + }, + { + "id": "2605.08442", + "title": "Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents", + "url": "https://arxiv.org/abs/2605.08442", + "published": "2026-05-08", + "updated": "2026-07-03", + "authors": [ + "Jun Wen Leong" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.08442", + "source": "arxiv", + "source_id": "arxiv:2605.08442", + "pdf_url": "https://arxiv.org/pdf/2605.08442", + "primary_query": "agent-safety" + }, + { + "id": "2605.06869", + "title": "Agentick: A Unified Benchmark for General Sequential Decision-Making Agents", + "url": "https://arxiv.org/abs/2605.06869", + "published": "2026-05-07", + "updated": "2026-05-12", + "authors": [ + "Roger Creus Castanyer", + "Pablo Samuel Castro", + "Glen Berseth" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.06869", + "source": "arxiv", + "source_id": "arxiv:2605.06869", + "pdf_url": "https://arxiv.org/pdf/2605.06869", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.03233", + "title": "Agentic and Generative AI for Open-Source Intelligence and Cyber Investigations: Taxonomy, Evaluation, Challenges, and Future Directions", + "url": "https://arxiv.org/abs/2607.03233", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Eduardo Almeida Palmieri", + "Mohamed Chahine Ghanem", + "Dipo Dunsin", + "Zubair Baig", + "Ed de Quincey", + "Kim-Kwang Raymond Choo" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.IR", + "cs.SI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "rag-agent", + "tool-use" + ], + "arxiv_id": "2607.03233", + "source": "arxiv", + "source_id": "arxiv:2607.03233", + "pdf_url": "https://arxiv.org/pdf/2607.03233", + "primary_query": "agentic-ai" + }, + { + "id": "2606.28061", + "title": "ToolPrivacyBench: Benchmarking Purpose-Bound Privacy in Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.28061", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Shijing Hu", + "Liang Liu", + "Zhu Meng", + "Zhicheng Zhao" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "function-calling", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.28061", + "source": "arxiv", + "source_id": "arxiv:2606.28061", + "pdf_url": "https://arxiv.org/pdf/2606.28061", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.17459", + "title": "Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation", + "url": "https://arxiv.org/abs/2606.17459", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Yuyang Dai", + "Xueqing Peng", + "Lingfei Qian", + "Zhuohan Xie" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "planning-agent" + ], + "arxiv_id": "2606.17459", + "source": "arxiv", + "source_id": "arxiv:2606.17459", + "pdf_url": "https://arxiv.org/pdf/2606.17459", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.16613", + "title": "CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies", + "url": "https://arxiv.org/abs/2606.16613", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Issa Sugiura", + "Daichi Hattori", + "Kazuo Araragi", + "Keita Ogawa", + "Shota Onose", + "Taro Makino", + "Teppei Usuki", + "Takashi Ishida" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2606.16613", + "source": "arxiv", + "source_id": "arxiv:2606.16613", + "pdf_url": "https://arxiv.org/pdf/2606.16613", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.12945", + "title": "Learning What to Remember: A Cognitively Grounded Multi-Factor Value Model for Agentic Memory", + "url": "https://arxiv.org/abs/2606.12945", + "published": "2026-06-11", + "updated": "2026-06-20", + "authors": [ + "Zhibao Chen", + "Qian Cheng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.12945", + "source": "arxiv", + "source_id": "arxiv:2606.12945", + "pdf_url": "https://arxiv.org/pdf/2606.12945", + "primary_query": "agent-memory" + }, + { + "id": "2606.06399", + "title": "CollabSim: A CSCW-Grounded Methodology for Investigating Collaborative Competence of LLM Agents through Controlled Multi-Agent Experiments", + "url": "https://arxiv.org/abs/2606.06399", + "published": "2026-06-04", + "updated": "2026-06-06", + "authors": [ + "Jiaju Chen", + "Bo Sun", + "Yuxuan Lu", + "Yun Wang", + "Dakuo Wang", + "Bingsheng Yao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.06399", + "source": "arxiv", + "source_id": "arxiv:2606.06399", + "pdf_url": "https://arxiv.org/pdf/2606.06399", + "primary_query": "planning-agent" + }, + { + "id": "2606.01199", + "title": "Can LLM Agents Sustain Long-Horizon Organizational Dynamics?", + "url": "https://arxiv.org/abs/2606.01199", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Xuancheng Zhu", + "Yang Yue", + "Shuaibing Wan", + "Zihan Dou", + "Xiaohan Zhang", + "Yongrui Liu", + "Guoshun Nan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "workflow-agent", + "world-model" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "language-agent", + "planning-agent" + ], + "arxiv_id": "2606.01199", + "source": "arxiv", + "source_id": "arxiv:2606.01199", + "pdf_url": "https://arxiv.org/pdf/2606.01199", + "primary_query": "language-agent" + }, + { + "id": "2605.29861", + "title": "Towards Verifiable Multimodal Deep Research: A Multi-Agent Harness for Interleaved Report Generation", + "url": "https://arxiv.org/abs/2605.29861", + "published": "2026-05-28", + "updated": "2026-06-03", + "authors": [ + "Chenghao Zhang", + "Guanting Dong", + "Yufan Liu", + "Tong Zhao", + "Xiaoxi Li", + "Zhicheng Dou" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.29861", + "source": "arxiv", + "source_id": "arxiv:2605.29861", + "pdf_url": "https://arxiv.org/pdf/2605.29861", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.18652", + "title": "MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents", + "url": "https://arxiv.org/abs/2605.18652", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Ziyun Zeng", + "Hang Hua", + "Bocheng Zou", + "Mu Cai", + "Rogerio Feris", + "Jiebo Luo" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.18652", + "source": "arxiv", + "source_id": "arxiv:2605.18652", + "pdf_url": "https://arxiv.org/pdf/2605.18652", + "primary_query": "agent-memory" + }, + { + "id": "2605.05704", + "title": "SafeHarbor: Hierarchical Memory-Augmented Guardrail for LLM Agent Safety", + "url": "https://arxiv.org/abs/2605.05704", + "published": "2026-05-07", + "updated": "2026-05-22", + "authors": [ + "Zhe Liu", + "Zonghao Ying", + "Wenxin Zhang", + "Quanchen Zou", + "Deyue Zhang", + "Dongdong Yang", + "Xiangzheng Zhang", + "Hao Peng" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "computer-use", + "memory", + "reasoning", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "autonomous-agent-llm" + ], + "arxiv_id": "2605.05704", + "source": "arxiv", + "source_id": "arxiv:2605.05704", + "pdf_url": "https://arxiv.org/pdf/2605.05704", + "primary_query": "agent-safety" + }, + { + "id": "2605.03242", + "title": "Enhancing Agent Safety Judgment: Controlled Benchmark Rewriting and Analogical Reasoning for Deceptive Out-of-Distribution Scenarios", + "url": "https://arxiv.org/abs/2605.03242", + "published": "2026-05-05", + "updated": "2026-05-05", + "authors": [ + "Zuoyu Zhang", + "Yancheng Zhu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.03242", + "source": "arxiv", + "source_id": "arxiv:2605.03242", + "pdf_url": "https://arxiv.org/pdf/2605.03242", + "primary_query": "agent-safety" + }, + { + "id": "2607.06118", + "title": "WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation", + "url": "https://arxiv.org/abs/2607.06118", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Wei Dong", + "Tianyu Fu", + "Zhe Yu", + "Hanning Wang", + "Anyang Su", + "Zhizhou Fang", + "Yuyang Chen", + "Shuo Wang", + "Minghui Wu", + "Ping Jiang", + "Zhen Lei", + "Chenxu Zhao" + ], + "categories": [ + "cs.CV", + "cs.MM" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "web-gui-agent" + ], + "arxiv_id": "2607.06118", + "source": "arxiv", + "source_id": "arxiv:2607.06118", + "pdf_url": "https://arxiv.org/pdf/2607.06118", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.05773", + "title": "Beyond Static Evaluation: Building Simulation Environments for Scalable Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2607.05773", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Akshay Arora", + "Ishan Nigam", + "Ashutosh Aggarwal", + "Shefali Bansal", + "Krishna Singh", + "Sweta Kumari", + "Nikhil Mittal", + "Shariq Farhan", + "Siddarth Malreddy" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "world-model" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "tool-use", + "web-gui-agent" + ], + "arxiv_id": "2607.05773", + "source": "arxiv", + "source_id": "arxiv:2607.05773", + "pdf_url": "https://arxiv.org/pdf/2607.05773", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.05456", + "title": "Prompt-to-Paper: Agentic AI System for Bioinformatics", + "url": "https://arxiv.org/abs/2607.05456", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Ramsha Kamran", + "Maheera Amjad", + "Zartasha Mustansar", + "Arsalan Shaukat", + "Salma Sherbaz", + "Muhammad U. S. Khan" + ], + "categories": [ + "cs.AI", + "cs.CL", + "q-bio.QM" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.05456", + "source": "arxiv", + "source_id": "arxiv:2607.05456", + "pdf_url": "https://arxiv.org/pdf/2607.05456", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04433", + "title": "Autonomous Information Seeking: A Roadmap for Agentic Recommender Systems", + "url": "https://arxiv.org/abs/2607.04433", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Xinyu Lin", + "Yashar Deldjoo", + "Sunhao Dai", + "Honghui Bao", + "Xiaopeng Ye", + "Fatemeh Nazary", + "Wenjie Wang", + "Tommaso Di Noia", + "Jun Xu", + "Tat-Seng Chua" + ], + "categories": [ + "cs.IR", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.04433", + "source": "arxiv", + "source_id": "arxiv:2607.04433", + "pdf_url": "https://arxiv.org/pdf/2607.04433", + "primary_query": "tool-use" + }, + { + "id": "2607.03601", + "title": "ArchEval: Measuring AI Agents as Computer Architects", + "url": "https://arxiv.org/abs/2607.03601", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Chenyu Wang", + "Zishen Wan", + "Jeffrey Ma", + "Shvetank Prakash", + "Zhenting Qi", + "Haebin Do", + "Andy Cheng", + "Arya Tschand", + "Jiahe Shi", + "Yilun Du", + "Vijay Janapa Reddi" + ], + "categories": [ + "cs.AR" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.03601", + "source": "arxiv", + "source_id": "arxiv:2607.03601", + "pdf_url": "https://arxiv.org/pdf/2607.03601", + "primary_query": "ai-agent" + }, + { + "id": "2607.02032", + "title": "PACE: A Proxy for Agentic Capability Evaluation", + "url": "https://arxiv.org/abs/2607.02032", + "published": "2026-07-02", + "updated": "2026-07-06", + "authors": [ + "Yueqi Song", + "Lintang Sutawika", + "Jiarui Liu", + "Lindia Tjuatja", + "Jiayi Geng", + "Yunze Xiao", + "Daniel Lee", + "Aditya Bharat Soni", + "Vincent Lo", + "Xiang Yue", + "Graham Neubig" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.02032", + "source": "arxiv", + "source_id": "arxiv:2607.02032", + "pdf_url": "https://arxiv.org/pdf/2607.02032", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.01641", + "title": "When Agents Do Not Stop: Uncovering Infinite Agentic Loops in LLM Agents", + "url": "https://arxiv.org/abs/2607.01641", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Xinyi Hou", + "Shenao Wang", + "Yanjie Zhao", + "Haoyu Wang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2607.01641", + "source": "arxiv", + "source_id": "arxiv:2607.01641", + "pdf_url": "https://arxiv.org/pdf/2607.01641", + "primary_query": "llm-agent" + }, + { + "id": "2606.30524", + "title": "The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems", + "url": "https://arxiv.org/abs/2606.30524", + "published": "2026-06-29", + "updated": "2026-07-01", + "authors": [ + "Abu Saleh", + "Tesfay Welegebreal Tesfay", + "Phuong T. Nguyen", + "Juri Di Rocco", + "Muhammad Umar Zeshan", + "Davide Di Ruscio" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "planning", + "rag" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "rag-agent" + ], + "arxiv_id": "2606.30524", + "source": "arxiv", + "source_id": "arxiv:2606.30524", + "pdf_url": "https://arxiv.org/pdf/2606.30524", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29537", + "title": "OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks", + "url": "https://arxiv.org/abs/2606.29537", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Mengqi Yuan", + "Zilong Zhou", + "Xinzhuang Xiong", + "Weiming Wu", + "Jiayang Sun", + "Jiamin Song", + "Kaiqian Cui", + "Bowen Wang", + "Haoyuan Wu", + "Yitong Li", + "Dunjie Lu", + "Haikong Lu", + "Qi Zhen", + "Xinyuan Wang", + "Jiaqi Deng", + "Yuhao Yang", + "Cheng Chen", + "Boyuan Zheng", + "Alex Su", + "Xiao Yu", + "Hao Zou", + "Saaket Agashe", + "Xing Han Lu", + "Manpreet Kaur", + "Zhengyang Qi", + "Vincent Sunn Chen", + "Frederic Sala", + "Dayiheng Liu", + "Junyang Lin", + "Zhou Yu", + "Yu Su", + "Siva Reddy", + "Xin Eric Wang", + "Peng Qi", + "Tianbao Xie", + "Tao Yu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.29537", + "source": "arxiv", + "source_id": "arxiv:2606.29537", + "pdf_url": "https://arxiv.org/pdf/2606.29537", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.28925", + "title": "Multi-Agent Routing as Set-Valued Prediction: A WildChat Benchmark and Cost-Aware Evaluation", + "url": "https://arxiv.org/abs/2606.28925", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Ananto Nayan Bala", + "Faisal Muhammad Shah" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.IR", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28925", + "source": "arxiv", + "source_id": "arxiv:2606.28925", + "pdf_url": "https://arxiv.org/pdf/2606.28925", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.26511", + "title": "Temporal Validity in Retrieval Memory: Eliminating Stale-Fact Errors for AI Agents over Evolving Knowledge", + "url": "https://arxiv.org/abs/2606.26511", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Neeraj Yadav" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.ET", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2606.26511", + "source": "arxiv", + "source_id": "arxiv:2606.26511", + "pdf_url": "https://arxiv.org/pdf/2606.26511", + "primary_query": "ai-agent" + }, + { + "id": "2606.22557", + "title": "MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop", + "url": "https://arxiv.org/abs/2606.22557", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Yikun Fu", + "Bowen Fu", + "Zhenyu Wu", + "Shuang Cheng", + "Xiaowei Sun", + "Bowen Yang", + "Zehao Li", + "Yibo Zhao", + "Zichen Ding", + "Zhoumianze Liu", + "Shijie Wang", + "Biqing Qi", + "Bowen Zhou" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "ai-agent", + "web-gui-agent" + ], + "arxiv_id": "2606.22557", + "source": "arxiv", + "source_id": "arxiv:2606.22557", + "pdf_url": "https://arxiv.org/pdf/2606.22557", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.21877", + "title": "AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems", + "url": "https://arxiv.org/abs/2606.21877", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Srimonti Dutta", + "Akshata Kishore Moharir" + ], + "categories": [ + "cs.AI", + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "rag-agent", + "tool-use" + ], + "arxiv_id": "2606.21877", + "source": "arxiv", + "source_id": "arxiv:2606.21877", + "pdf_url": "https://arxiv.org/pdf/2606.21877", + "primary_query": "agentic-ai" + }, + { + "id": "2606.16802", + "title": "LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control", + "url": "https://arxiv.org/abs/2606.16802", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Anqi Zou", + "Han Deng", + "Chengyu Zhang", + "Junquan Hu", + "Yu Wang", + "Yuxiang Xing", + "Aokai Zhang", + "Hanling Zhang", + "Zhaoyang Liu", + "Ben Fei", + "Zhihui Wang", + "Wanli Ouyang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.16802", + "source": "arxiv", + "source_id": "arxiv:2606.16802", + "pdf_url": "https://arxiv.org/pdf/2606.16802", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13994", + "title": "Hidden in Plain Sight: Benchmarking Agent Safety Against Decomposition Attacks with DECOMPBENCH", + "url": "https://arxiv.org/abs/2606.13994", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Vikhyath Kothamasu", + "Virginia Smith", + "Chhavi Yadav" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "tool-use" + ], + "arxiv_id": "2606.13994", + "source": "arxiv", + "source_id": "arxiv:2606.13994", + "pdf_url": "https://arxiv.org/pdf/2606.13994", + "primary_query": "agent-safety" + }, + { + "id": "2606.04990", + "title": "From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents", + "url": "https://arxiv.org/abs/2606.04990", + "published": "2026-06-03", + "updated": "2026-06-28", + "authors": [ + "Yiqi Wang", + "Jiaqi Zhang", + "Taotao Cai", + "Zirui Liu", + "Qingqiang Sun", + "Zequn Sun", + "Zhangkai Wu", + "Manqing Dong", + "Mingkai Zheng", + "Xuefei Yin", + "Yanming Zhu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.04990", + "source": "arxiv", + "source_id": "arxiv:2606.04990", + "pdf_url": "https://arxiv.org/pdf/2606.04990", + "primary_query": "planning-agent" + }, + { + "id": "2606.02461", + "title": "AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents", + "url": "https://arxiv.org/abs/2606.02461", + "published": "2026-06-01", + "updated": "2026-06-02", + "authors": [ + "Yiheng Shu", + "Bernal Jiménez Gutiérrez", + "Saisri Padmaja Jonnalagedda", + "Yuguang Yao", + "Huan Sun", + "Yu Su" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.02461", + "source": "arxiv", + "source_id": "arxiv:2606.02461", + "pdf_url": "https://arxiv.org/pdf/2606.02461", + "primary_query": "language-agent" + }, + { + "id": "2606.01385", + "title": "Bridging Requirements and Architecture: Multi-Agent Orchestration with External Knowledge and Hierarchical Memory", + "url": "https://arxiv.org/abs/2606.01385", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Ruiyin Li", + "Yiran Zhang", + "Xiyu Zhou", + "Yangxiao Cai", + "Peng Liang", + "Weisong Sun", + "Jifeng Xuan", + "Zhi Jin", + "Yang Liu" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "multi-agent", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.01385", + "source": "arxiv", + "source_id": "arxiv:2606.01385", + "pdf_url": "https://arxiv.org/pdf/2606.01385", + "primary_query": "rag-agent" + }, + { + "id": "2606.00756", + "title": "CoMIC: Collaborative Memory and Insights Circulation for Long-Horizon LLM Agents in Cloud-Edge Systems", + "url": "https://arxiv.org/abs/2606.00756", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Yannan Wang", + "Longli Yang", + "Zhen Liu", + "Abhishek Kumar", + "Carsten Maple" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.00756", + "source": "arxiv", + "source_id": "arxiv:2606.00756", + "pdf_url": "https://arxiv.org/pdf/2606.00756", + "primary_query": "planning-agent" + }, + { + "id": "2605.20315", + "title": "Mix-Quant: Quantized Prefilling, Precise Decoding for Agentic LLMs", + "url": "https://arxiv.org/abs/2605.20315", + "published": "2026-05-19", + "updated": "2026-05-19", + "authors": [ + "Haiquan Lu", + "Zigeng Chen", + "Gongfan Fang", + "Xinyin Ma", + "Xinchao Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.20315", + "source": "arxiv", + "source_id": "arxiv:2605.20315", + "pdf_url": "https://arxiv.org/pdf/2605.20315", + "primary_query": "planning-agent" + }, + { + "id": "2605.14498", + "title": "GroupMemBench: Benchmarking LLM Agent Memory in Multi-Party Conversations", + "url": "https://arxiv.org/abs/2605.14498", + "published": "2026-05-14", + "updated": "2026-05-16", + "authors": [ + "Jingbo Yang", + "Kwei-Herng Lai", + "Xiaowen Wang", + "Shiyu Chang", + "Yaar Harari", + "Evgeniy Gabrilovich" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.14498", + "source": "arxiv", + "source_id": "arxiv:2605.14498", + "pdf_url": "https://arxiv.org/pdf/2605.14498", + "primary_query": "agent-memory" + }, + { + "id": "2605.11633", + "title": "Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations", + "url": "https://arxiv.org/abs/2605.11633", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Junjue Wang", + "Weihao Xuan", + "Heli Qi", + "Pengyu Dai", + "Kunyi Liu", + "Hongruixuan Chen", + "Zhuo Zheng", + "Junshi Xia", + "Stefano Ermon", + "Naoto Yokoya" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.11633", + "source": "arxiv", + "source_id": "arxiv:2605.11633", + "pdf_url": "https://arxiv.org/pdf/2605.11633", + "primary_query": "planning-agent" + }, + { + "id": "2605.06812", + "title": "Towards Security-Auditable LLM Agents: A Unified Graph Representation", + "url": "https://arxiv.org/abs/2605.06812", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Chaofan Li", + "Lyuye Zhang", + "Jintao Zhai", + "Siyue Feng", + "Xichun Yang", + "Huahao Wang", + "Shihan Dou", + "Yu Ji", + "Yutao Hu", + "Yueming Wu", + "Yang Liu", + "Deqing Zou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.06812", + "source": "arxiv", + "source_id": "arxiv:2605.06812", + "pdf_url": "https://arxiv.org/pdf/2605.06812", + "primary_query": "agent-safety" + }, + { + "id": "2602.08412", + "title": "From Assistant to Double Agent: Formalizing and Benchmarking Attacks on OpenClaw for Personalized Local AI Agent", + "url": "https://arxiv.org/abs/2602.08412", + "published": "2026-02-09", + "updated": "2026-02-11", + "authors": [ + "Yuhang Wang", + "Feiming Xu", + "Zheng Lin", + "Guangyu He", + "Yuzhe Huang", + "Haichang Gao", + "Zhenxing Niu", + "Shiguo Lian", + "Zhaoxiang Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.08412", + "source": "arxiv", + "source_id": "arxiv:2602.08412", + "pdf_url": "https://arxiv.org/pdf/2602.08412", + "primary_query": "agent-safety" + }, + { + "id": "2508.07575", + "title": "MCPToolBench++: A Large Scale AI Agent Model Context Protocol MCP Tool Use Benchmark", + "url": "https://arxiv.org/abs/2508.07575", + "published": "2025-08-11", + "updated": "2025-08-11", + "authors": [ + "Shiqing Fan", + "Xichen Ding", + "Liang Zhang", + "Linjian Mo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.07575", + "source": "arxiv", + "source_id": "arxiv:2508.07575", + "pdf_url": "https://arxiv.org/pdf/2508.07575", + "primary_query": "function-calling" + }, + { + "id": "2607.05318", + "title": "PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems", + "url": "https://arxiv.org/abs/2607.05318", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Shubham Gupta", + "Nazanin Mohammadi Sepahvand", + "Abhinav Kumar", + "Cem Subakan", + "Spandana Gella", + "Pierre-André Noël", + "Perouz Taslakian", + "Eugene Bagdasarian", + "Valentina Zantedeschi" + ], + "categories": [ + "cs.MA", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.05318", + "source": "arxiv", + "source_id": "arxiv:2607.05318", + "pdf_url": "https://arxiv.org/pdf/2607.05318", + "primary_query": "llm-agent" + }, + { + "id": "2607.04391", + "title": "Memory-Orchestrated Semantic System (MOSS): An Auditable Agentic Memory Architecture", + "url": "https://arxiv.org/abs/2607.04391", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Serge Lacasse", + "Jérémie Hatier", + "Alex Baker" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2607.04391", + "source": "arxiv", + "source_id": "arxiv:2607.04391", + "pdf_url": "https://arxiv.org/pdf/2607.04391", + "primary_query": "agent-memory" + }, + { + "id": "2607.03953", + "title": "The Remarkable Effectiveness of Providing AI Agents with Natural Language Tools: A Replication Study Validating NLT Performance Across 14 Models", + "url": "https://arxiv.org/abs/2607.03953", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Alexander Somma", + "Isabelle Plante", + "Fred Premji" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.03953", + "source": "arxiv", + "source_id": "arxiv:2607.03953", + "pdf_url": "https://arxiv.org/pdf/2607.03953", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03726", + "title": "SelfMem: Self-Optimizing Memory for AI Agents", + "url": "https://arxiv.org/abs/2607.03726", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Shu Yang", + "Junchao Wu", + "Derek F. Wong", + "Di Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "ai-agent", + "tool-use" + ], + "arxiv_id": "2607.03726", + "source": "arxiv", + "source_id": "arxiv:2607.03726", + "pdf_url": "https://arxiv.org/pdf/2607.03726", + "primary_query": "agent-memory" + }, + { + "id": "2607.01766", + "title": "SimWorlds: A Multi-Agent System for Dynamic 3D Scene Creation", + "url": "https://arxiv.org/abs/2607.01766", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Chunjiang Liu", + "Xiaoyuan Wang", + "Haoyu Chen", + "Yizhou Zhao", + "Ming-Hsuan Yang", + "László A. Jeni" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.01766", + "source": "arxiv", + "source_id": "arxiv:2607.01766", + "pdf_url": "https://arxiv.org/pdf/2607.01766", + "primary_query": "llm-agent" + }, + { + "id": "2607.00454", + "title": "Agri-SAGE: Simulation-Grounded Multi-Agent LLM for Context-Aware Agricultural Advisory Generation", + "url": "https://arxiv.org/abs/2607.00454", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Vedant Balasubramaniam", + "Geetha Charan", + "Manojkumar Patil", + "Rohit P Suresh", + "V Priyanka", + "Kodur Sai Vinay Sathvik", + "Y. Narahari" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "world-model" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00454", + "source": "arxiv", + "source_id": "arxiv:2607.00454", + "pdf_url": "https://arxiv.org/pdf/2607.00454", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31073", + "title": "MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning", + "url": "https://arxiv.org/abs/2606.31073", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Sheng Zhang", + "Qinglin Li", + "Yuechao Zang", + "Xueqin Huang", + "Yijia Fu", + "Cheng Zhu" + ], + "categories": [ + "cs.AI", + "cs.MA", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use", + "world-model" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.31073", + "source": "arxiv", + "source_id": "arxiv:2606.31073", + "pdf_url": "https://arxiv.org/pdf/2606.31073", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.31179", + "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents", + "url": "https://arxiv.org/abs/2606.31179", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Qianchu Liu", + "Sheng Zhang", + "Guanghui Qin", + "Jeya Maria Jose Valanarasu", + "Maximilian Rokuss", + "Mingyu Lu", + "Timothy Ossowski", + "Juan Manuel Zambrano Chaves", + "Cliff Wong", + "Peniel Argaw", + "Yashna Hasija", + "Mu Wei", + "Wen-wai Yim", + "Qin Liu", + "Zilin Jing", + "Jason Entenmann", + "Naoto Usuyama", + "Tristan Naumann", + "Hoifung Poon" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "ai-agent" + ], + "arxiv_id": "2606.31179", + "source": "arxiv", + "source_id": "arxiv:2606.31179", + "pdf_url": "https://arxiv.org/pdf/2606.31179", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.31612", + "title": "What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States", + "url": "https://arxiv.org/abs/2606.31612", + "published": "2026-06-30", + "updated": "2026-07-02", + "authors": [ + "Chen Liu", + "Ling Chen", + "Hanzhang Zhou", + "Xu Zhang", + "Quyu Kong", + "Panrong Tong", + "Wenhao Wang", + "Xin Yu", + "Steven Hoi", + "Yue Wang" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "web-gui-agent" + ], + "arxiv_id": "2606.31612", + "source": "arxiv", + "source_id": "arxiv:2606.31612", + "pdf_url": "https://arxiv.org/pdf/2606.31612", + "primary_query": "agent-memory" + }, + { + "id": "2606.30906", + "title": "Investigating Multi-Agent Deliberation in Law", + "url": "https://arxiv.org/abs/2606.30906", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Cor Steging", + "Ludi van Leeuwen", + "Tadeusz Zbiegień" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.30906", + "source": "arxiv", + "source_id": "arxiv:2606.30906", + "pdf_url": "https://arxiv.org/pdf/2606.30906", + "primary_query": "agentic-ai" + }, + { + "id": "2606.30949", + "title": "AgRefactor: Self-Evolving Agentic Workflow for HLS Compatibility and Performance", + "url": "https://arxiv.org/abs/2606.30949", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yang Zou", + "Zijian Ding", + "Yizhou Sun", + "Jason Cong" + ], + "categories": [ + "cs.AI", + "cs.AR" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.30949", + "source": "arxiv", + "source_id": "arxiv:2606.30949", + "pdf_url": "https://arxiv.org/pdf/2606.30949", + "primary_query": "agentic-ai" + }, + { + "id": "2606.29116", + "title": "Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem", + "url": "https://arxiv.org/abs/2606.29116", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Yutian Tang", + "Yuming Zhou", + "Huaming Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2606.29116", + "source": "arxiv", + "source_id": "arxiv:2606.29116", + "pdf_url": "https://arxiv.org/pdf/2606.29116", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24535", + "title": "Governed Shared Memory for Multi-Agent LLM Systems", + "url": "https://arxiv.org/abs/2606.24535", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yanki Margalit", + "Nurit Cohen-Inger", + "Erni Avram", + "Ran Taig", + "Oded Margalit" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "multi-agent-llm" + ], + "arxiv_id": "2606.24535", + "source": "arxiv", + "source_id": "arxiv:2606.24535", + "pdf_url": "https://arxiv.org/pdf/2606.24535", + "primary_query": "agent-memory" + }, + { + "id": "2606.24820", + "title": "SHERLOC: Structured Diagnostic Localization for Code Repair Agents", + "url": "https://arxiv.org/abs/2606.24820", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Hovhannes Tamoyan", + "Sean Narenthiran", + "Erik Arakelyan", + "Mira Mezini", + "Boris Ginsburg" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "multi-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.24820", + "source": "arxiv", + "source_id": "arxiv:2606.24820", + "pdf_url": "https://arxiv.org/pdf/2606.24820", + "primary_query": "coding-agent" + }, + { + "id": "2606.22263", + "title": "Revelio: Cost-Efficient Agentic Memory Safety Vulnerability Detection For Repository-Scale Codebases", + "url": "https://arxiv.org/abs/2606.22263", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Yiwei Hou", + "Hao Wang", + "Muxi Lyu", + "Marius Momeu", + "Eric Nguyen", + "Taige Yang", + "Koushik Sen", + "Dawn Song", + "David Wagner" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.MA", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "coding-agent" + ], + "arxiv_id": "2606.22263", + "source": "arxiv", + "source_id": "arxiv:2606.22263", + "pdf_url": "https://arxiv.org/pdf/2606.22263", + "primary_query": "agent-memory" + }, + { + "id": "2606.21627", + "title": "Counsel: A Meta-Evaluation Dataset for Agentic Tasks", + "url": "https://arxiv.org/abs/2606.21627", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Sashank Pisupati", + "Henry Broomfield", + "Eujeong Choi", + "Antonia Calvi", + "Charlie Wang", + "Roman Engeler", + "Max Bartolo", + "Patrick Lewis" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2606.21627", + "source": "arxiv", + "source_id": "arxiv:2606.21627", + "pdf_url": "https://arxiv.org/pdf/2606.21627", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.16871", + "title": "Human-on-the-Bridge: Scalable Evaluation for AI Agents", + "url": "https://arxiv.org/abs/2606.16871", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Fouad Bousetouane" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.16871", + "source": "arxiv", + "source_id": "arxiv:2606.16871", + "pdf_url": "https://arxiv.org/pdf/2606.16871", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.17246", + "title": "GeoDisaster: Benchmarking Orchestrated Agents for Operational Disaster Geo-Intelligence", + "url": "https://arxiv.org/abs/2606.17246", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Maram Hasan", + "Aman Verma", + "Savitra Roy", + "Hariseetharam Gunduboina", + "Daksh Jain", + "Muhammad Haris Khan", + "Subhasis Chaudhuri", + "Biplab Banerjee" + ], + "categories": [ + "cs.CV", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.17246", + "source": "arxiv", + "source_id": "arxiv:2606.17246", + "pdf_url": "https://arxiv.org/pdf/2606.17246", + "primary_query": "tool-use" + }, + { + "id": "2606.17114", + "title": "An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios", + "url": "https://arxiv.org/abs/2606.17114", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Hankyul Baek", + "Jaewon Noh", + "Sang Seo", + "Yongsu Kim", + "Gabriel Waikin Loh Matienzo", + "Young Il Kim", + "Ee Wei Seah", + "Akriti Vij" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "tool-use" + ], + "arxiv_id": "2606.17114", + "source": "arxiv", + "source_id": "arxiv:2606.17114", + "pdf_url": "https://arxiv.org/pdf/2606.17114", + "primary_query": "agent-safety" + }, + { + "id": "2606.16420", + "title": "Transferable Self-Evolving Playbooks for Agentic Security Auditing", + "url": "https://arxiv.org/abs/2606.16420", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Ziyue Wang", + "Cheuk Wang Maurice Ng", + "Chenchen Yu", + "Strick Sheng", + "Kaihua Qin", + "Liyi Zhou" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "tool-use" + ], + "arxiv_id": "2606.16420", + "source": "arxiv", + "source_id": "arxiv:2606.16420", + "pdf_url": "https://arxiv.org/pdf/2606.16420", + "primary_query": "agent-safety" + }, + { + "id": "2606.14790", + "title": "XFlow: An Executable Protocol Programming System for Reliable Multi-Agent Workflows", + "url": "https://arxiv.org/abs/2606.14790", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Hanqi Li", + "Jing Peng", + "Zijian Wang", + "Lu Chen", + "Kai Yu" + ], + "categories": [ + "cs.PL", + "cs.AI" + ], + "topics": [ + "coding-agent", + "memory", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.14790", + "source": "arxiv", + "source_id": "arxiv:2606.14790", + "pdf_url": "https://arxiv.org/pdf/2606.14790", + "primary_query": "tool-use" + }, + { + "id": "2606.08531", + "title": "VESTA: A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents", + "url": "https://arxiv.org/abs/2606.08531", + "published": "2026-06-07", + "updated": "2026-06-07", + "authors": [ + "Lu Jia", + "Haibo Tong", + "Feifei Zhao", + "Jindong Li", + "Dongqi Liang", + "Ping Wu", + "Qian Zhang", + "Yi Zeng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.08531", + "source": "arxiv", + "source_id": "arxiv:2606.08531", + "pdf_url": "https://arxiv.org/pdf/2606.08531", + "primary_query": "agent-safety" + }, + { + "id": "2606.07402", + "title": "M$^3$Exam: Benchmarking Multimodal Memory for Realistic User-Agent Interactions", + "url": "https://arxiv.org/abs/2606.07402", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Zhengjun Huang", + "Wenxuan Liu", + "Zhoujin Tian", + "Wei Chen", + "Junle Chen", + "Yuqian Wu", + "Fangyuan Zhang", + "Qintian Guo", + "Xiaofang Zhou" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.07402", + "source": "arxiv", + "source_id": "arxiv:2606.07402", + "pdf_url": "https://arxiv.org/pdf/2606.07402", + "primary_query": "language-agent" + }, + { + "id": "2606.04780", + "title": "PersonaTree: Structured Lifecycle Memory for Person Understanding in LLM Agents", + "url": "https://arxiv.org/abs/2606.04780", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Yubo Hou", + "Jingwei Song", + "Hongbo Zhang", + "Zhisheng Chen", + "Bang Xiao", + "Tao Wan", + "Zengchang Qin" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04780", + "source": "arxiv", + "source_id": "arxiv:2606.04780", + "pdf_url": "https://arxiv.org/pdf/2606.04780", + "primary_query": "agent-memory" + }, + { + "id": "2606.03657", + "title": "Diagnosing Knowledge Gaps in LLM Tool Use: An Agentic Benchmark for Novel API Acquisition", + "url": "https://arxiv.org/abs/2606.03657", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Jinnuo Liu", + "Yue Peng", + "Jinhan Niu", + "Hongyi Wen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.03657", + "source": "arxiv", + "source_id": "arxiv:2606.03657", + "pdf_url": "https://arxiv.org/pdf/2606.03657", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.02109", + "title": "BADGER: Bridging Agentic and Deterministic Evaluation for Generative Enterprise Reasoning", + "url": "https://arxiv.org/abs/2606.02109", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Shannon Serrao", + "Soumitra Chatterjee", + "Dorina Strori", + "Abhishek Sharma", + "Nathan Miller" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.02109", + "source": "arxiv", + "source_id": "arxiv:2606.02109", + "pdf_url": "https://arxiv.org/pdf/2606.02109", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.01416", + "title": "Self-Healing Agentic Orchestrators for Reliable Tool-Augmented Large Language Model Systems", + "url": "https://arxiv.org/abs/2606.01416", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Rahul Suresh Babu", + "Adarsh Agrawal" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.01416", + "source": "arxiv", + "source_id": "arxiv:2606.01416", + "pdf_url": "https://arxiv.org/pdf/2606.01416", + "primary_query": "planning-agent" + }, + { + "id": "2606.00610", + "title": "MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation", + "url": "https://arxiv.org/abs/2606.00610", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Chuanjie Wu", + "Zhishang Xiang", + "Yunbo Tang", + "Zerui Chen", + "Qinggang Zhang", + "Jinsong Su" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.00610", + "source": "arxiv", + "source_id": "arxiv:2606.00610", + "pdf_url": "https://arxiv.org/pdf/2606.00610", + "primary_query": "rag-agent" + }, + { + "id": "2605.30883", + "title": "TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking", + "url": "https://arxiv.org/abs/2605.30883", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Churui Zeng", + "Weiwei Qi", + "Kedong Xiu", + "Tianhang Zheng", + "Chaochao Lu", + "Liang He", + "Zhan Qin", + "Kui Ren" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.30883", + "source": "arxiv", + "source_id": "arxiv:2605.30883", + "pdf_url": "https://arxiv.org/pdf/2605.30883", + "primary_query": "planning-agent" + }, + { + "id": "2605.30090", + "title": "DirectorBench: Diagnosing Long-Form Video Generation with Personalized Multi-Agent Evaluation", + "url": "https://arxiv.org/abs/2605.30090", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Jiamin Chen", + "Qianben Chen", + "Jiawen Zhang", + "Yidi Wu", + "Yuchen Li", + "Xiaokun Zhang", + "Wangchunshu Zhou", + "Chen Ma" + ], + "categories": [ + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.30090", + "source": "arxiv", + "source_id": "arxiv:2605.30090", + "pdf_url": "https://arxiv.org/pdf/2605.30090", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.27134", + "title": "Scaling, Benchmarking, and Reasoning of Vision-Language Agents for Mobile GUI Navigation", + "url": "https://arxiv.org/abs/2605.27134", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Heng Qu", + "Yike Liu", + "Renren Jin", + "Wenzong Zhang", + "Pengzhi Gao", + "Wei Liu", + "Jian Luan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.27134", + "source": "arxiv", + "source_id": "arxiv:2605.27134", + "pdf_url": "https://arxiv.org/pdf/2605.27134", + "primary_query": "language-agent" + }, + { + "id": "2605.22643", + "title": "Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety", + "url": "https://arxiv.org/abs/2605.22643", + "published": "2026-05-21", + "updated": "2026-05-22", + "authors": [ + "Piercosma Bisconti", + "Matteo Prandi", + "Federico Pierucci", + "Federico Sartore", + "Enrico Panai", + "Laura Caroli", + "Yue Zhu", + "Adam Leon Smith", + "Luca Nannini", + "Marcello Galisai", + "Susanna Cifani", + "Francesco Giarrusso", + "Marcantonio Bracale Syrnikov", + "Daniele Nardi" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.22643", + "source": "arxiv", + "source_id": "arxiv:2605.22643", + "pdf_url": "https://arxiv.org/pdf/2605.22643", + "primary_query": "agent-safety" + }, + { + "id": "2605.15040", + "title": "Orchard: An Open-Source Agentic Modeling Framework", + "url": "https://arxiv.org/abs/2605.15040", + "published": "2026-05-14", + "updated": "2026-05-21", + "authors": [ + "Baolin Peng", + "Wenlin Yao", + "Qianhui Wu", + "Hao Cheng", + "Xiao Yu", + "Rui Yang", + "Tao Ge", + "Alessandro Sordoni", + "Xingdi Yuan", + "Yelong Shen", + "Pengcheng He", + "Tong Zhang", + "Zhou Yu", + "Jianfeng Gao" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.15040", + "source": "arxiv", + "source_id": "arxiv:2605.15040", + "pdf_url": "https://arxiv.org/pdf/2605.15040", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.13542", + "title": "RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation", + "url": "https://arxiv.org/abs/2605.13542", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Chengzhi Shen", + "Weixiang Shen", + "Tobias Susetzky", + "Chen", + "Chen", + "Jun Li", + "Yuyuan Liu", + "Xuepeng Zhang", + "Zhenyu Gong", + "Daniel Rueckert", + "Jiazhen Pan" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.13542", + "source": "arxiv", + "source_id": "arxiv:2605.13542", + "pdf_url": "https://arxiv.org/pdf/2605.13542", + "primary_query": "agent-memory" + }, + { + "id": "2605.12015", + "title": "SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces", + "url": "https://arxiv.org/abs/2605.12015", + "published": "2026-05-12", + "updated": "2026-05-27", + "authors": [ + "Chang Jin", + "An Wang", + "Zeming Wei", + "Kai Wang", + "Biaojie Zeng", + "Qiaosheng Zhang", + "Chao Yang", + "Jingjing Qu", + "Xia Hu", + "Xingcheng Xu" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.12015", + "source": "arxiv", + "source_id": "arxiv:2605.12015", + "pdf_url": "https://arxiv.org/pdf/2605.12015", + "primary_query": "agent-safety" + }, + { + "id": "2605.08374", + "title": "MemQ: Integrating Q-Learning into Self-Evolving Memory Agents over Provenance DAGs", + "url": "https://arxiv.org/abs/2605.08374", + "published": "2026-05-08", + "updated": "2026-05-14", + "authors": [ + "Junwei Liao", + "Haoting Shi", + "Ruiwen Zhou", + "Jiaqian Wang", + "Shengtao Zhang", + "Wei Zhang", + "Ying Wen", + "Zhiyu Li", + "Feiyu Xiong", + "Bo Tang", + "Weinan Zhang", + "Muning Wen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.08374", + "source": "arxiv", + "source_id": "arxiv:2605.08374", + "pdf_url": "https://arxiv.org/pdf/2605.08374", + "primary_query": "function-calling" + }, + { + "id": "2605.15206", + "title": "AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices", + "url": "https://arxiv.org/abs/2605.15206", + "published": "2026-05-01", + "updated": "2026-05-01", + "authors": [ + "Dzung Pham", + "Kleomenis Katevas", + "Ali Shahin Shamsabadi", + "Hamed Haddadi" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.15206", + "source": "arxiv", + "source_id": "arxiv:2605.15206", + "pdf_url": "https://arxiv.org/pdf/2605.15206", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.16282", + "title": "Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents", + "url": "https://arxiv.org/abs/2605.16282", + "published": "2026-04-11", + "updated": "2026-04-11", + "authors": [ + "Miles Q. Li", + "Benjamin C. M. Fung", + "Boyang Li", + "Heba Ismail", + "Farkhund Iqbal" + ], + "categories": [ + "cs.CY", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.16282", + "source": "arxiv", + "source_id": "arxiv:2605.16282", + "pdf_url": "https://arxiv.org/pdf/2605.16282", + "primary_query": "agent-safety" + }, + { + "id": "2604.02022", + "title": "ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis", + "url": "https://arxiv.org/abs/2604.02022", + "published": "2026-04-02", + "updated": "2026-05-13", + "authors": [ + "Yu Li", + "Haoyu Luo", + "Yuejin Xie", + "Yuqian Fu", + "Zhonghao Yang", + "Shuai Shao", + "Qihan Ren", + "Wanying Qu", + "Yanwei Fu", + "Yujiu Yang", + "Jing Shao", + "Xia Hu", + "Dongrui Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.02022", + "source": "arxiv", + "source_id": "arxiv:2604.02022", + "pdf_url": "https://arxiv.org/pdf/2604.02022", + "primary_query": "agent-safety" + }, + { + "id": "2603.16734", + "title": "Differential Harm Propensity in Personalized LLM Agents: The Curious Case of Mental Health Disclosure", + "url": "https://arxiv.org/abs/2603.16734", + "published": "2026-03-17", + "updated": "2026-03-17", + "authors": [ + "Caglar Yildirim" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.16734", + "source": "arxiv", + "source_id": "arxiv:2603.16734", + "pdf_url": "https://arxiv.org/pdf/2603.16734", + "primary_query": "agent-safety" + }, + { + "id": "2509.20998", + "title": "CORE: Full-Path Evaluation of LLM Agents Beyond Final State", + "url": "https://arxiv.org/abs/2509.20998", + "published": "2025-09-25", + "updated": "2025-09-25", + "authors": [ + "Panagiotis Michelakis", + "Yiannis Hadjiyiannis", + "Dimitrios Stamoulis" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "world-model" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.20998", + "source": "arxiv", + "source_id": "arxiv:2509.20998", + "pdf_url": "https://arxiv.org/pdf/2509.20998", + "primary_query": "function-calling" + }, + { + "id": "2607.05174", + "title": "AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments", + "url": "https://arxiv.org/abs/2607.05174", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Zhiheng Xi", + "Dingwen Yang", + "Jiaqi Liu", + "Jixuan Huang", + "Honglin Guo", + "Baodai Huang", + "Tinggang Chen", + "Qi Zhang", + "Zhonghang Lu", + "Chenyu Liu", + "Jiajun Sun", + "Jiazheng Zhang", + "Dingwei Zhu", + "Xin Guo", + "Junzhe Wang", + "Zhihao Zhang", + "Yuming Yang", + "Junjie Ye", + "Minghe Gao", + "Dongrui Liu", + "Jiaming Ji", + "Guohao Li", + "Tao Gui", + "Qi Zhang", + "Xuanjing Huang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent", + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.05174", + "source": "arxiv", + "source_id": "arxiv:2607.05174", + "pdf_url": "https://arxiv.org/pdf/2607.05174", + "primary_query": "language-agent" + }, + { + "id": "2607.05029", + "title": "Your Agent's Memories Are Not Its Own: Forged Reasoning Attacks on LLM Agent Memory and Defenses", + "url": "https://arxiv.org/abs/2607.05029", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Neeraj Karamchandani", + "Piyush Nagasubramaniam", + "Sencun Zhu", + "Dinghao Wu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "llm-agent" + ], + "arxiv_id": "2607.05029", + "source": "arxiv", + "source_id": "arxiv:2607.05029", + "pdf_url": "https://arxiv.org/pdf/2607.05029", + "primary_query": "agent-memory" + }, + { + "id": "2607.05120", + "title": "Agent Data Injection Attacks are Realistic Threats to AI Agents", + "url": "https://arxiv.org/abs/2607.05120", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Woohyuk Choi", + "Juhee Kim", + "Taehyun Kang", + "Jihyeon Jeong", + "Luyi Xing", + "Byoungyoung Lee" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "ai-agent", + "coding-agent", + "web-gui-agent" + ], + "arxiv_id": "2607.05120", + "source": "arxiv", + "source_id": "arxiv:2607.05120", + "pdf_url": "https://arxiv.org/pdf/2607.05120", + "primary_query": "agent-safety" + }, + { + "id": "2607.05202", + "title": "EvoAgentBench: Benchmarking Agent Self-Evolution via Ability Transfer", + "url": "https://arxiv.org/abs/2607.05202", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Xingze Gao", + "Chuanrui Hu", + "Hongda Chen", + "Pengfei Yao", + "Zhao Wang", + "Yi Bai", + "Zhengwei Wu", + "Yunyun Han", + "Xiaofeng Cong", + "Jie Gui", + "Yafeng Deng", + "Teng Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "planning", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2607.05202", + "source": "arxiv", + "source_id": "arxiv:2607.05202", + "pdf_url": "https://arxiv.org/pdf/2607.05202", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.04395", + "title": "NKI-Agent: Domain-Specific Fine-Tuning and Agentic Tool Use for Neuron Kernel Generation", + "url": "https://arxiv.org/abs/2607.04395", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Junjie Tang", + "Jun Huan", + "Hao Zhou", + "Yuhao Zhang", + "Lin Wang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.04395", + "source": "arxiv", + "source_id": "arxiv:2607.04395", + "pdf_url": "https://arxiv.org/pdf/2607.04395", + "primary_query": "tool-use" + }, + { + "id": "2607.03510", + "title": "CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI", + "url": "https://arxiv.org/abs/2607.03510", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Roopam W. Sure" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.03510", + "source": "arxiv", + "source_id": "arxiv:2607.03510", + "pdf_url": "https://arxiv.org/pdf/2607.03510", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02684", + "title": "Can Coding Agents Implement Missed Compiler Optimizations? Evaluating LLM Agents on LLVM Peephole Optimizations", + "url": "https://arxiv.org/abs/2607.02684", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Hongxu Xu", + "Chunhao Liao", + "Xintong Zhou", + "Chengnian Sun" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.02684", + "source": "arxiv", + "source_id": "arxiv:2607.02684", + "pdf_url": "https://arxiv.org/pdf/2607.02684", + "primary_query": "coding-agent" + }, + { + "id": "2607.02507", + "title": "What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates", + "url": "https://arxiv.org/abs/2607.02507", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Arman Ghaffarizadeh", + "Danyal Mohaddes", + "Aliakbar Izadkhah", + "Shahriar Noroozizadeh" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.02507", + "source": "arxiv", + "source_id": "arxiv:2607.02507", + "pdf_url": "https://arxiv.org/pdf/2607.02507", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.30986", + "title": "The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows", + "url": "https://arxiv.org/abs/2606.30986", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Canhui Liu" + ], + "categories": [ + "cs.CY", + "cs.HC", + "cs.MA", + "econ.GN" + ], + "topics": [ + "memory", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent" + ], + "arxiv_id": "2606.30986", + "source": "arxiv", + "source_id": "arxiv:2606.30986", + "pdf_url": "https://arxiv.org/pdf/2606.30986", + "primary_query": "agentic-ai" + }, + { + "id": "2606.30639", + "title": "Self-Evolving World Models for LLM Agent Planning", + "url": "https://arxiv.org/abs/2606.30639", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Xuan Zhang", + "Wenxuan Zhang", + "See-Kiong Ng", + "Yang Deng" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.30639", + "source": "arxiv", + "source_id": "arxiv:2606.30639", + "pdf_url": "https://arxiv.org/pdf/2606.30639", + "primary_query": "llm-agent" + }, + { + "id": "2606.29771", + "title": "CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents", + "url": "https://arxiv.org/abs/2606.29771", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Bo Qu", + "Mingguang Chen" + ], + "categories": [ + "cs.AI", + "cs.LG", + "q-fin.CP", + "q-fin.PM" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29771", + "source": "arxiv", + "source_id": "arxiv:2606.29771", + "pdf_url": "https://arxiv.org/pdf/2606.29771", + "primary_query": "llm-agent" + }, + { + "id": "2607.00041", + "title": "ATM: CID-Brokered Pre-Write Admission for Multi-Agent Code Co-Synthesis", + "url": "https://arxiv.org/abs/2607.00041", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Eagl Huang" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "planning", + "rag" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "multi-agent-llm" + ], + "arxiv_id": "2607.00041", + "source": "arxiv", + "source_id": "arxiv:2607.00041", + "pdf_url": "https://arxiv.org/pdf/2607.00041", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.29774", + "title": "Analytic Concept-Centric Memory for Agentic Embodied Manipulation", + "url": "https://arxiv.org/abs/2606.29774", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Mingyang Sun", + "Xiujian Liang", + "Jiude Wei", + "Qichen He", + "Donglin Wang", + "Cewu Lu", + "Jianhua Sun" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.29774", + "source": "arxiv", + "source_id": "arxiv:2606.29774", + "pdf_url": "https://arxiv.org/pdf/2606.29774", + "primary_query": "agent-memory" + }, + { + "id": "2606.29193", + "title": "A Multi-Dataset Benchmark for Evaluating LLM Agents in Microservice Failure Diagnosis", + "url": "https://arxiv.org/abs/2606.29193", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Yuanhong Cai", + "Xiaohui Nie", + "Kanglin Yin", + "Changhua Pei", + "Yongqian Sun", + "Shenglin Zhang", + "Haibin Liu", + "Guiyang Liu", + "Xidao Wen", + "Fang Situ", + "Dan Pei" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29193", + "source": "arxiv", + "source_id": "arxiv:2606.29193", + "pdf_url": "https://arxiv.org/pdf/2606.29193", + "primary_query": "llm-agent" + }, + { + "id": "2606.29030", + "title": "Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering", + "url": "https://arxiv.org/abs/2606.29030", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Shahnewaz Karim Sakib", + "Anindya Bijoy Das" + ], + "categories": [ + "cs.AI", + "cs.ET" + ], + "topics": [ + "memory", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.29030", + "source": "arxiv", + "source_id": "arxiv:2606.29030", + "pdf_url": "https://arxiv.org/pdf/2606.29030", + "primary_query": "ai-agent" + }, + { + "id": "2606.28692", + "title": "An AI agent for treatment reasoning over a biomedical tool universe", + "url": "https://arxiv.org/abs/2606.28692", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Shanghua Gao", + "Ayush Noori", + "Richard Zhu", + "Curtis Ginder", + "Zhenglun Kong", + "Xiaorui Su", + "Justin Kauffman", + "Benjamin S. Glicksberg", + "Joshua Lampert", + "Ankit Sakhuja", + "Ashwin Sawant", + "ATHENA-R1 Evaluation Consortium", + "David A. Clifton", + "Noa Dagan", + "Ran Balicer", + "Marinka Zitnik" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "tool-use" + ], + "arxiv_id": "2606.28692", + "source": "arxiv", + "source_id": "arxiv:2606.28692", + "pdf_url": "https://arxiv.org/pdf/2606.28692", + "primary_query": "ai-agent" + }, + { + "id": "2606.27806", + "title": "Agent vs. Parametric World Models: Hybrid Planning for Reliable Language Agents", + "url": "https://arxiv.org/abs/2606.27806", + "published": "2026-06-26", + "updated": "2026-07-05", + "authors": [ + "Xinyuan Song", + "Zekun Cai" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.27806", + "source": "arxiv", + "source_id": "arxiv:2606.27806", + "pdf_url": "https://arxiv.org/pdf/2606.27806", + "primary_query": "language-agent" + }, + { + "id": "2606.28467", + "title": "An Agentic AI Pipeline for Appliance-Level Energy Anomaly Detection and LLM-Driven Recommendations", + "url": "https://arxiv.org/abs/2606.28467", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Dihia Falouz", + "Aida Douaibia", + "Amine Bechar", + "Youssef Elmir", + "Abbes Amira", + "Adel Oulefki" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "rag-agent" + ], + "arxiv_id": "2606.28467", + "source": "arxiv", + "source_id": "arxiv:2606.28467", + "pdf_url": "https://arxiv.org/pdf/2606.28467", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26960", + "title": "Toward Agentic SysAdmin: Rethinking System Administration with AI Agents", + "url": "https://arxiv.org/abs/2606.26960", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Gianmaria Frigo", + "Davide Saladino", + "Alberto Castagnaro", + "Francesco Marchiori", + "Denis Donadel", + "Luca Pajola", + "Mauro Conti" + ], + "categories": [ + "cs.NI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.26960", + "source": "arxiv", + "source_id": "arxiv:2606.26960", + "pdf_url": "https://arxiv.org/pdf/2606.26960", + "primary_query": "ai-agent" + }, + { + "id": "2606.26346", + "title": "How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?", + "url": "https://arxiv.org/abs/2606.26346", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "David Akinpelu", + "Akintonde Abbas", + "Rereloluwa Alimi", + "Ayodeji Lana" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.26346", + "source": "arxiv", + "source_id": "arxiv:2606.26346", + "pdf_url": "https://arxiv.org/pdf/2606.26346", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.26403", + "title": "ProfileFoundry: A Synthetic Person-Object Substrate for Privacy, Memory, and Tool-Use Evaluation in LLM Agent", + "url": "https://arxiv.org/abs/2606.26403", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Sriram Selvam", + "Anneswa Ghosh" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.26403", + "source": "arxiv", + "source_id": "arxiv:2606.26403", + "pdf_url": "https://arxiv.org/pdf/2606.26403", + "primary_query": "tool-use" + }, + { + "id": "2606.25161", + "title": "TRUSTMEM: Learning Trustworthy Memory Consolidation for LLM Agents with Long-Term Memory", + "url": "https://arxiv.org/abs/2606.25161", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Tianyu Yang", + "Sudipta Paul", + "Vijay Srinivasan", + "Vivek Kulkarni", + "Srinivas Chappidi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.25161", + "source": "arxiv", + "source_id": "arxiv:2606.25161", + "pdf_url": "https://arxiv.org/pdf/2606.25161", + "primary_query": "agent-memory" + }, + { + "id": "2606.24626", + "title": "SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation", + "url": "https://arxiv.org/abs/2606.24626", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Chenyang Zhu", + "Jiayu Yao", + "Kushal Chawla", + "Youbing Yin", + "Nathan Wolfe", + "Pengshan Cai", + "Jingyu Wu", + "Spencer Hong", + "Sangwoo Cho", + "Shi-Xiong Zhang", + "Daben Liu", + "Sambit Sahu", + "Erin Babinsky" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "multi-agent-llm" + ], + "arxiv_id": "2606.24626", + "source": "arxiv", + "source_id": "arxiv:2606.24626", + "pdf_url": "https://arxiv.org/pdf/2606.24626", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.23991", + "title": "Critique of Agent Model", + "url": "https://arxiv.org/abs/2606.23991", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Eric Xing", + "Mingkai Deng", + "Jinyu Hou" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.MA", + "cs.RO" + ], + "topics": [ + "agent-safety", + "coding-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2606.23991", + "source": "arxiv", + "source_id": "arxiv:2606.23991", + "pdf_url": "https://arxiv.org/pdf/2606.23991", + "primary_query": "agentic-ai" + }, + { + "id": "2606.22844", + "title": "RaMem: Contextual Reinstatement for Long-term Agentic Memory", + "url": "https://arxiv.org/abs/2606.22844", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Wei Yang", + "Bryce Kan", + "Shixuan Li", + "Li Li", + "Yuehan Qin", + "Jiate Li", + "Paul Bogdan", + "Jesse Thomason" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.22844", + "source": "arxiv", + "source_id": "arxiv:2606.22844", + "pdf_url": "https://arxiv.org/pdf/2606.22844", + "primary_query": "agent-memory" + }, + { + "id": "2606.23565", + "title": "HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory", + "url": "https://arxiv.org/abs/2606.23565", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Xiaolin Zhou", + "Liu Liu", + "Tingyang Xiao", + "Wei Feng", + "Fa Fu", + "Xinrui Meng", + "Xinjie Wang", + "Jialiang Han", + "Boyang Yu", + "Yun Du", + "Wei Sui", + "Zhizhong Su" + ], + "categories": [ + "cs.RO", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.23565", + "source": "arxiv", + "source_id": "arxiv:2606.23565", + "pdf_url": "https://arxiv.org/pdf/2606.23565", + "primary_query": "planning-agent" + }, + { + "id": "2606.22678", + "title": "RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents", + "url": "https://arxiv.org/abs/2606.22678", + "published": "2026-06-21", + "updated": "2026-06-29", + "authors": [ + "Meher Bhaskar Madiraju", + "Meher Sai Preetam Madiraju" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.22678", + "source": "arxiv", + "source_id": "arxiv:2606.22678", + "pdf_url": "https://arxiv.org/pdf/2606.22678", + "primary_query": "coding-agent" + }, + { + "id": "2606.22417", + "title": "Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent", + "url": "https://arxiv.org/abs/2606.22417", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Ishaan Bhola", + "Adithyan Krishnan", + "Sravanth Kurmala", + "Mukunda NS" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.22417", + "source": "arxiv", + "source_id": "arxiv:2606.22417", + "pdf_url": "https://arxiv.org/pdf/2606.22417", + "primary_query": "coding-agent" + }, + { + "id": "2606.21129", + "title": "AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents", + "url": "https://arxiv.org/abs/2606.21129", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Zhen Zhao", + "Yu Zhang", + "Yanpeng Zhu", + "Jia Wang", + "Songqiao Tao", + "Xin Cheng", + "Jiexin Gao" + ], + "categories": [ + "cs.CR", + "cs.OS" + ], + "topics": [ + "agent-safety", + "planning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "autonomous-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.21129", + "source": "arxiv", + "source_id": "arxiv:2606.21129", + "pdf_url": "https://arxiv.org/pdf/2606.21129", + "primary_query": "ai-agent" + }, + { + "id": "2606.21649", + "title": "EvoEmbedding: Evolvable Representations for Long-Context Retrieval and Agentic Memory", + "url": "https://arxiv.org/abs/2606.21649", + "published": "2026-06-19", + "updated": "2026-06-25", + "authors": [ + "Chang Nie", + "Chaoyou Fu", + "Junlan Feng", + "Caifeng Shan" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "agentic-ai", + "rag-agent" + ], + "arxiv_id": "2606.21649", + "source": "arxiv", + "source_id": "arxiv:2606.21649", + "pdf_url": "https://arxiv.org/pdf/2606.21649", + "primary_query": "agent-memory" + }, + { + "id": "2606.20950", + "title": "Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering", + "url": "https://arxiv.org/abs/2606.20950", + "published": "2026-06-18", + "updated": "2026-07-02", + "authors": [ + "Sergei Trashchenkov" + ], + "categories": [ + "cs.AI", + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "ai-agent", + "tool-use" + ], + "arxiv_id": "2606.20950", + "source": "arxiv", + "source_id": "arxiv:2606.20950", + "pdf_url": "https://arxiv.org/pdf/2606.20950", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.19704", + "title": "Beyond Static Leaderboards: Predictive Validity for the Evaluation of LLM Agents", + "url": "https://arxiv.org/abs/2606.19704", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Dhaval C. Patel", + "Kaoutar El Maghraoui", + "Shuxin Lin", + "Yusheng Li", + "Tianjun Feng", + "Chun-Yi Tsai", + "Yihan Sun", + "Wei Alexander Xin", + "Akshat Bhandari", + "Tanisha Rathod", + "Aaron Fan", + "Sanskruti Vijay Shejwal", + "Tomas Pasiecznik", + "Sagar Chethan Kumar", + "Tanmay Agarwal", + "Rohith Kanathur", + "Sam Colman", + "Amaan Sheikh", + "Dev Bahl", + "Ann Li", + "Krish Veera", + "Alimurtaza Mustafa Merchant", + "Shambhawi Baswaraj Bhure", + "Sajal Kumar Goyla", + "Chengrui Li", + "Kirthana Natarajan", + "Rui Li", + "Thomas Ajai", + "Rujing Li", + "Vivek G. Iyer", + "Sanjaii Vijayakumar", + "Yitong Bai", + "Ayal Yakobe", + "Darief Maes", + "Yassine Jebbouri", + "Tianyang Xu", + "Thai Quoc On", + "Vera Mazeeva", + "Winston Li", + "Yuval Shemla", + "Yeshitha Bhuvanesh", + "Rushin Bhatt", + "Siddharth Chethan Gowda", + "Alisha Vinod", + "Caroline Cahill", + "Shriya Aishani Rachakonda", + "Yunfeng Chen", + "Aryaman Agrawal", + "Aman Upganlawar", + "Mao Le Jonathan Ang", + "Yubin Sally Go", + "Madhav Rajkondawar", + "Yang-Jung Chen", + "Trisha Maturi", + "Ananya Kapoor", + "Andrew Li", + "Shrey Arora", + "Mana Abbaszadeh", + "Shen Li", + "Charles Xu", + "Byeolah Kwon" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.19704", + "source": "arxiv", + "source_id": "arxiv:2606.19704", + "pdf_url": "https://arxiv.org/pdf/2606.19704", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.20512", + "title": "Probe-and-Refine Tuning of Repository Guidance for Coding Agents", + "url": "https://arxiv.org/abs/2606.20512", + "published": "2026-06-18", + "updated": "2026-06-19", + "authors": [ + "Asa Shepard", + "Jeannie Albrecht" + ], + "categories": [ + "cs.SE", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "tool-use" + ], + "arxiv_id": "2606.20512", + "source": "arxiv", + "source_id": "arxiv:2606.20512", + "pdf_url": "https://arxiv.org/pdf/2606.20512", + "primary_query": "coding-agent" + }, + { + "id": "2606.18829", + "title": "GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents", + "url": "https://arxiv.org/abs/2606.18829", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Zhe Ren", + "Yibo Yang", + "Yimeng Chen", + "Zijun Zhao", + "Benshuo Fu", + "Zhihao Shu", + "Bingjie Zhang", + "Yangyang Xu", + "Dandan Guo", + "Shuicheng Yan" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.18829", + "source": "arxiv", + "source_id": "arxiv:2606.18829", + "pdf_url": "https://arxiv.org/pdf/2606.18829", + "primary_query": "agent-memory" + }, + { + "id": "2606.18356", + "title": "SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.18356", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Yuchuan Tian", + "Mengyu Zheng", + "Haocheng Mei", + "Ye Yuan", + "Chao Xu", + "Xinghao Chen", + "Hanting Chen", + "Yu Wang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "tool-use" + ], + "arxiv_id": "2606.18356", + "source": "arxiv", + "source_id": "arxiv:2606.18356", + "pdf_url": "https://arxiv.org/pdf/2606.18356", + "primary_query": "agent-safety" + }, + { + "id": "2606.16774", + "title": "OpenClaw-Skill: Collective Skill Tree Search for Agentic Large Language Models", + "url": "https://arxiv.org/abs/2606.16774", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Tianyi Lin", + "Chuanyu Sun", + "Jingyi Zhang", + "Changxu Wei", + "Huanjin Yao", + "Shunyu Liu", + "Xikun Zhang", + "Liu Liu", + "Jiaxing Huang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent", + "tool-use" + ], + "arxiv_id": "2606.16774", + "source": "arxiv", + "source_id": "arxiv:2606.16774", + "pdf_url": "https://arxiv.org/pdf/2606.16774", + "primary_query": "planning-agent" + }, + { + "id": "2606.15862", + "title": "RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments", + "url": "https://arxiv.org/abs/2606.15862", + "published": "2026-06-14", + "updated": "2026-06-19", + "authors": [ + "Linghua Zhang", + "Jun Wang", + "Jingtong Wu", + "Zhisong Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.15862", + "source": "arxiv", + "source_id": "arxiv:2606.15862", + "pdf_url": "https://arxiv.org/pdf/2606.15862", + "primary_query": "tool-use" + }, + { + "id": "2606.12586", + "title": "Beyond Attack Success Rate: Examining Trigger Leakage in Vision-Language Agentic Systems", + "url": "https://arxiv.org/abs/2606.12586", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Jiamin Chang", + "Salil Kanhere", + "Piotr Koniusz", + "Jason", + "Xue", + "Hammond Pearce" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent", + "tool-use" + ], + "arxiv_id": "2606.12586", + "source": "arxiv", + "source_id": "arxiv:2606.12586", + "pdf_url": "https://arxiv.org/pdf/2606.12586", + "primary_query": "language-agent" + }, + { + "id": "2606.10507", + "title": "HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning", + "url": "https://arxiv.org/abs/2606.10507", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Juncheng Diao", + "Zhicong Lu", + "Peiguang Li", + "Yongwei Zhou", + "Changyuan Tian", + "Qingbin Li", + "Rongxiang Weng", + "Jingang Wang", + "Xunliang Cai" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2606.10507", + "source": "arxiv", + "source_id": "arxiv:2606.10507", + "pdf_url": "https://arxiv.org/pdf/2606.10507", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.11042", + "title": "Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields", + "url": "https://arxiv.org/abs/2606.11042", + "published": "2026-06-09", + "updated": "2026-06-11", + "authors": [ + "Liya Zhu", + "Jingzhe Ding", + "Jian Zhang", + "Jianbo Xue", + "Shihao Liang", + "Ge Zhang", + "Yi Zhu", + "Duju Zeng", + "Xiang Gao", + "Qingshui Gu", + "Mailun Gao", + "Huimin Che", + "Yan Zhao", + "Peiheng Zhou", + "Haojun Wang", + "Chaobo Xian", + "Lili Le", + "Chi Wu", + "Yiwei Liu", + "Shengda Long", + "Jiale Yang", + "Fangzhi Xu", + "Sijin Wu", + "Haodong Duan", + "Chao He", + "Zhaojian Li", + "Minchao Wang", + "Huan Zhou", + "Jiani Hou", + "Chuqian Yu", + "Weiran Shi", + "Hongwan Gao", + "Jiamin Chen", + "Guanhong Chen", + "Tingqin Luo", + "Kaiyuan Zhang", + "Zhixin Yao", + "Qing Hua", + "Yuhao Jiang", + "Jin Chen", + "Pu Chen", + "Zhenyu Hu", + "Xingyu Li", + "Zhengxuan Jiang", + "Meng Cao", + "Tianfeng Long", + "Haozhe Wang", + "Mingzhang Wang", + "Yichen Zhang", + "Yiming Dai", + "Chenchen Zhang", + "Jiaying Wang", + "Xinying Liu", + "Xingzu Liu", + "Lingling Zhang", + "Xinjie Chen", + "Yujia Qin", + "Wangchunshu Zhou", + "Zhiyong Wu", + "Yang Liu", + "Jiaheng Liu", + "Lei Zhang", + "Shen Yan", + "Wenhao Huang", + "Zaiyuan Wang", + "Xiaolong Chang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.11042", + "source": "arxiv", + "source_id": "arxiv:2606.11042", + "pdf_url": "https://arxiv.org/pdf/2606.11042", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09483", + "title": "Memory Beyond Recall: A Dual-Process Cognitive Memory System for Self-Evolving LLM Agents", + "url": "https://arxiv.org/abs/2606.09483", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Tianxiang Fei", + "Mingyang Song", + "Mao Zheng", + "Xiang Yu" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.09483", + "source": "arxiv", + "source_id": "arxiv:2606.09483", + "pdf_url": "https://arxiv.org/pdf/2606.09483", + "primary_query": "agent-memory" + }, + { + "id": "2606.07314", + "title": "QBugLM: An Agentic Benchmarking Framework for LLM-based Quantum Software Debugging", + "url": "https://arxiv.org/abs/2606.07314", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "An B. B. Pham", + "Hoa T. Nguyen", + "Muhammad Usman" + ], + "categories": [ + "cs.SE", + "cs.ET", + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "reasoning", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.07314", + "source": "arxiv", + "source_id": "arxiv:2606.07314", + "pdf_url": "https://arxiv.org/pdf/2606.07314", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.05684", + "title": "AdaMEM: Test-Time Adaptive Memory for Language Agents", + "url": "https://arxiv.org/abs/2606.05684", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Yunxiang Zhang", + "Yiheng Li", + "Ali Payani", + "Lu Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "language-agent" + ], + "arxiv_id": "2606.05684", + "source": "arxiv", + "source_id": "arxiv:2606.05684", + "pdf_url": "https://arxiv.org/pdf/2606.05684", + "primary_query": "agent-memory" + }, + { + "id": "2606.06448", + "title": "Agent Memory: Characterization and System Implications of Stateful Long-Horizon Workloads", + "url": "https://arxiv.org/abs/2606.06448", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Yasmine Omri", + "Ziyu Gan", + "Zachary Broveak", + "Robin Geens", + "Zexue He", + "Alex Pentland", + "Marian Verhelst", + "Tsachy Weissman", + "Thierry Tambe" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.06448", + "source": "arxiv", + "source_id": "arxiv:2606.06448", + "pdf_url": "https://arxiv.org/pdf/2606.06448", + "primary_query": "agent-memory" + }, + { + "id": "2606.04874", + "title": "Agent Planning Benchmark: A Diagnostic Framework for Planning Capabilities in LLM Agents", + "url": "https://arxiv.org/abs/2606.04874", + "published": "2026-06-03", + "updated": "2026-06-05", + "authors": [ + "Haoyu Sun", + "Wenxuan Wang", + "Mingyang Song", + "Jujie He", + "Weinan Zhang", + "Yang Liu", + "Yang Yang", + "Yu Cheng" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "planning-agent" + ], + "arxiv_id": "2606.04874", + "source": "arxiv", + "source_id": "arxiv:2606.04874", + "pdf_url": "https://arxiv.org/pdf/2606.04874", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.28349", + "title": "HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning", + "url": "https://arxiv.org/abs/2606.28349", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Zeju Li", + "Ziyang Zheng", + "Yizhou Zhou", + "Qiang Xu" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.28349", + "source": "arxiv", + "source_id": "arxiv:2606.28349", + "pdf_url": "https://arxiv.org/pdf/2606.28349", + "primary_query": "agent-memory" + }, + { + "id": "2606.04315", + "title": "Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline", + "url": "https://arxiv.org/abs/2606.04315", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Zhikai Chen", + "Jialiang Gu", + "Junyu Yin", + "Xianxuan Long", + "Shenglai Zeng", + "Xiaoze Liu", + "Kai Guo", + "Keren Zhou", + "Jiliang Tang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04315", + "source": "arxiv", + "source_id": "arxiv:2606.04315", + "pdf_url": "https://arxiv.org/pdf/2606.04315", + "primary_query": "agent-memory" + }, + { + "id": "2606.03374", + "title": "eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents", + "url": "https://arxiv.org/abs/2606.03374", + "published": "2026-06-02", + "updated": "2026-06-22", + "authors": [ + "A. Haroon Rasheed", + "Maria Kabtoul" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2606.03374", + "source": "arxiv", + "source_id": "arxiv:2606.03374", + "pdf_url": "https://arxiv.org/pdf/2606.03374", + "primary_query": "agent-memory" + }, + { + "id": "2606.02372", + "title": "COMAP: Co-Evolving World Models and Agent Policies for LLM Agents", + "url": "https://arxiv.org/abs/2606.02372", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Youwei Liu", + "Jian Wang", + "Hanlin Wang", + "Wenjie Li" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent", + "planning-agent" + ], + "arxiv_id": "2606.02372", + "source": "arxiv", + "source_id": "arxiv:2606.02372", + "pdf_url": "https://arxiv.org/pdf/2606.02372", + "primary_query": "language-agent" + }, + { + "id": "2606.01613", + "title": "TechRAG: Evidence-Gated Multimodal Agentic RAG for Technical Literature Reasoning", + "url": "https://arxiv.org/abs/2606.01613", + "published": "2026-06-01", + "updated": "2026-06-13", + "authors": [ + "Kanwar Bharat Singh" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.01613", + "source": "arxiv", + "source_id": "arxiv:2606.01613", + "pdf_url": "https://arxiv.org/pdf/2606.01613", + "primary_query": "rag-agent" + }, + { + "id": "2606.00939", + "title": "FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation", + "url": "https://arxiv.org/abs/2606.00939", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Chao Peter Yang", + "Zixiao Tan", + "Kaisen Yao", + "Ziyu Zhou", + "Eleanor Jiang", + "Michael Wu" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.00939", + "source": "arxiv", + "source_id": "arxiv:2606.00939", + "pdf_url": "https://arxiv.org/pdf/2606.00939", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.30690", + "title": "ElasticMem: Latent Memory as a Learnable Resource for LLM Agents", + "url": "https://arxiv.org/abs/2605.30690", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Tao Feng", + "Chongrui Ye", + "Tianyang Luo", + "Jingjun Xu", + "Xueqiang Xu", + "Haozhen Zhang", + "Ge Liu", + "Jiaxuan You" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.30690", + "source": "arxiv", + "source_id": "arxiv:2605.30690", + "pdf_url": "https://arxiv.org/pdf/2605.30690", + "primary_query": "planning-agent" + }, + { + "id": "2606.20629", + "title": "Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams", + "url": "https://arxiv.org/abs/2606.20629", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Yinsicheng Jiang", + "Liang Cheng", + "Yeqi Huang", + "Yufan Zhao", + "Zhan Lu", + "Li Dong", + "Wenda Li", + "Edoardo Ponti", + "Luo Mai" + ], + "categories": [ + "cs.MA", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.20629", + "source": "arxiv", + "source_id": "arxiv:2606.20629", + "pdf_url": "https://arxiv.org/pdf/2606.20629", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.30604", + "title": "An Organization-Scoped LLM Agent Runtime Architecture for Regulated Cybersecurity Operations", + "url": "https://arxiv.org/abs/2605.30604", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "George Fatouros", + "Georgios Makridis", + "George Kousiouris", + "John Soldatos", + "Dimosthenis Kyriazis" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.30604", + "source": "arxiv", + "source_id": "arxiv:2605.30604", + "pdf_url": "https://arxiv.org/pdf/2605.30604", + "primary_query": "planning-agent" + }, + { + "id": "2605.27240", + "title": "ENPMR-Bench: Benchmarking Proactive Memory Retrieval for Emotional Support Agents", + "url": "https://arxiv.org/abs/2605.27240", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Xing Fu", + "Yulin Hu", + "Mengtong Ji", + "Haozhen Li", + "Yixin Sun", + "Weixiang Zhao", + "Yanyan Zhao", + "Bing Qin" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.27240", + "source": "arxiv", + "source_id": "arxiv:2605.27240", + "pdf_url": "https://arxiv.org/pdf/2605.27240", + "primary_query": "language-agent" + }, + { + "id": "2605.25200", + "title": "GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning", + "url": "https://arxiv.org/abs/2605.25200", + "published": "2026-05-24", + "updated": "2026-06-03", + "authors": [ + "Xiang Cheng", + "Yulan Hu", + "Lulu Zheng", + "Zheng Pan", + "Xin Li", + "Yong Liu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.25200", + "source": "arxiv", + "source_id": "arxiv:2605.25200", + "pdf_url": "https://arxiv.org/pdf/2605.25200", + "primary_query": "planning-agent" + }, + { + "id": "2605.24220", + "title": "Polar: Agentic RL on Any Harness at Scale", + "url": "https://arxiv.org/abs/2605.24220", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Binfeng Xu", + "Hao Zhang", + "Shaokun Zhang", + "Songyang Han", + "Mingjie Liu", + "Jian Hu", + "Shizhe Diao", + "Zhenghui Jin", + "Yunheng Zou", + "Michael Demoret", + "Jan Kautz", + "Yi Dong" + ], + "categories": [ + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.24220", + "source": "arxiv", + "source_id": "arxiv:2605.24220", + "pdf_url": "https://arxiv.org/pdf/2605.24220", + "primary_query": "language-agent" + }, + { + "id": "2605.23574", + "title": "Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2605.23574", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Yuandao Cai", + "Yuzhang Zhu", + "Liyou Gao", + "Wensheng Tang", + "Shengchao Qin" + ], + "categories": [ + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.23574", + "source": "arxiv", + "source_id": "arxiv:2605.23574", + "pdf_url": "https://arxiv.org/pdf/2605.23574", + "primary_query": "language-agent" + }, + { + "id": "2605.24069", + "title": "When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents", + "url": "https://arxiv.org/abs/2605.24069", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Shi Liu", + "Xuehai Tang", + "Xikang Yang", + "Liang Lin", + "Biyu Zhou", + "Wenjie Xiao", + "Wantao Liu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.24069", + "source": "arxiv", + "source_id": "arxiv:2605.24069", + "pdf_url": "https://arxiv.org/pdf/2605.24069", + "primary_query": "planning-agent" + }, + { + "id": "2605.24216", + "title": "Agent-ToM: Learning to Monitor Autonomous LLM Agents via Theory-of-Mind Reasoning", + "url": "https://arxiv.org/abs/2605.24216", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Nesreen K. Ahmed", + "Nima Nafisi" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.24216", + "source": "arxiv", + "source_id": "arxiv:2605.24216", + "pdf_url": "https://arxiv.org/pdf/2605.24216", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.18421", + "title": "EvoMemBench: Benchmarking Agent Memory from a Self-Evolving Perspective", + "url": "https://arxiv.org/abs/2605.18421", + "published": "2026-05-18", + "updated": "2026-06-15", + "authors": [ + "Yuyao Wang", + "Zhongjian Zhang", + "Mo Chi", + "Kaichi Yu", + "Yuhan Li", + "Miao Peng", + "Bing Tong", + "Chen Zhang", + "Yan Zhou", + "Jia Li" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "planning-agent" + ], + "arxiv_id": "2605.18421", + "source": "arxiv", + "source_id": "arxiv:2605.18421", + "pdf_url": "https://arxiv.org/pdf/2605.18421", + "primary_query": "agent-memory" + }, + { + "id": "2605.10779", + "title": "LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments", + "url": "https://arxiv.org/abs/2605.10779", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Chiyu Zhang", + "Huiqin Yang", + "Bendong Jiang", + "Xiaolei Zhang", + "Yiran Zhao", + "Ruyi Chen", + "Lu Zhou", + "Xiaogang Xu", + "Jiafei Wu", + "Liming Fang", + "Zhe Liu" + ], + "categories": [ + "cs.CR", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.10779", + "source": "arxiv", + "source_id": "arxiv:2605.10779", + "pdf_url": "https://arxiv.org/pdf/2605.10779", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.03312", + "title": "MemFlow: Intent-Driven Memory Orchestration for Small Language Model Agents", + "url": "https://arxiv.org/abs/2605.03312", + "published": "2026-05-05", + "updated": "2026-05-05", + "authors": [ + "Jiayi Chen", + "Yingcong Li", + "Guiling Wang" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.03312", + "source": "arxiv", + "source_id": "arxiv:2605.03312", + "pdf_url": "https://arxiv.org/pdf/2605.03312", + "primary_query": "language-agent" + }, + { + "id": "2605.01101", + "title": "Virtual Speech Therapist: A Clinician-in-the-Loop AI Speech Therapy Agent for Personalized and Supervised Therapy", + "url": "https://arxiv.org/abs/2605.01101", + "published": "2026-05-01", + "updated": "2026-06-15", + "authors": [ + "Shakeel Sheikh", + "Patrick Marmaroli", + "MD Sahidullah", + "Slim Ouni", + "Fabrice Hirsch", + "Goncalo Leal", + "Bjorn W Schuller" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.SD", + "eess.AS" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.01101", + "source": "arxiv", + "source_id": "arxiv:2605.01101", + "pdf_url": "https://arxiv.org/pdf/2605.01101", + "primary_query": "planning-agent" + }, + { + "id": "2604.19844", + "title": "If you're waiting for a sign... that might not be it! Mitigating Trust Boundary Confusion from Visual Injections on Vision-Language Agentic Systems", + "url": "https://arxiv.org/abs/2604.19844", + "published": "2026-04-21", + "updated": "2026-04-21", + "authors": [ + "Jiamin Chang", + "Minhui Xue", + "Ruoxi Sun", + "Shuchao Pang", + "Salil S. Kanhere", + "Hammond Pearce" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "multi-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.19844", + "source": "arxiv", + "source_id": "arxiv:2604.19844", + "pdf_url": "https://arxiv.org/pdf/2604.19844", + "primary_query": "language-agent" + }, + { + "id": "2604.18658", + "title": "Owner-Harm: A Missing Threat Model for AI Agent Safety", + "url": "https://arxiv.org/abs/2604.18658", + "published": "2026-04-20", + "updated": "2026-04-20", + "authors": [ + "Dongcheng Zhang", + "Yiqing Jiang" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.18658", + "source": "arxiv", + "source_id": "arxiv:2604.18658", + "pdf_url": "https://arxiv.org/pdf/2604.18658", + "primary_query": "agent-safety" + }, + { + "id": "2604.17562", + "title": "SafeAgent: A Runtime Protection Architecture for Agentic Systems", + "url": "https://arxiv.org/abs/2604.17562", + "published": "2026-04-19", + "updated": "2026-04-19", + "authors": [ + "Hailin Liu", + "Eugene Ilyushin", + "Jie Ni", + "Min Zhu" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.17562", + "source": "arxiv", + "source_id": "arxiv:2604.17562", + "pdf_url": "https://arxiv.org/pdf/2604.17562", + "primary_query": "agent-safety" + }, + { + "id": "2603.00623", + "title": "TraceSIR: A Multi-Agent Framework for Structured Analysis and Reporting of Agentic Execution Traces", + "url": "https://arxiv.org/abs/2603.00623", + "published": "2026-02-28", + "updated": "2026-02-28", + "authors": [ + "Shu-Xun Yang", + "Cunxiang Wang", + "Haoke Zhang", + "Wenbo Yu", + "Lindong Wu", + "Jiayi Gui", + "Dayong Yang", + "Yukuo Cen", + "Zhuoer Feng", + "Bosi Wen", + "Yidong Wang", + "Lucen Zhong", + "Jiamin Ren", + "Linfeng Zhang", + "Jie Tang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.00623", + "source": "arxiv", + "source_id": "arxiv:2603.00623", + "pdf_url": "https://arxiv.org/pdf/2603.00623", + "primary_query": "function-calling" + }, + { + "id": "2602.13530", + "title": "REMem: Reasoning with Episodic Memory in Language Agent", + "url": "https://arxiv.org/abs/2602.13530", + "published": "2026-02-13", + "updated": "2026-02-28", + "authors": [ + "Yiheng Shu", + "Saisri Padmaja Jonnalagedda", + "Xiang Gao", + "Bernal Jiménez Gutiérrez", + "Weijian Qi", + "Kamalika Das", + "Huan Sun", + "Yu Su" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.13530", + "source": "arxiv", + "source_id": "arxiv:2602.13530", + "pdf_url": "https://arxiv.org/pdf/2602.13530", + "primary_query": "language-agent" + }, + { + "id": "2510.03847", + "title": "Small Language Models for Agentic Systems: A Survey of Architectures, Capabilities, and Deployment Trade offs", + "url": "https://arxiv.org/abs/2510.03847", + "published": "2025-10-04", + "updated": "2025-10-04", + "authors": [ + "Raghav Sharma", + "Manan Mehta" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.03847", + "source": "arxiv", + "source_id": "arxiv:2510.03847", + "pdf_url": "https://arxiv.org/pdf/2510.03847", + "primary_query": "function-calling" + }, + { + "id": "2607.06157", + "title": "LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability", + "url": "https://arxiv.org/abs/2607.06157", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Chenxu Wang", + "Yongkun Yang", + "Boyuan Du", + "Shiwei Lin", + "Huaping Liu" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.06157", + "source": "arxiv", + "source_id": "arxiv:2607.06157", + "pdf_url": "https://arxiv.org/pdf/2607.06157", + "primary_query": "llm-agent" + }, + { + "id": "2607.04686", + "title": "ToolFailBench: Diagnosing Tool-Use Failures in LLM Agents", + "url": "https://arxiv.org/abs/2607.04686", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Harsh Soni" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.04686", + "source": "arxiv", + "source_id": "arxiv:2607.04686", + "pdf_url": "https://arxiv.org/pdf/2607.04686", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04240", + "title": "Biological Motifs for Agentic Control", + "url": "https://arxiv.org/abs/2607.04240", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Bogdan Banu" + ], + "categories": [ + "cs.AI", + "q-bio.CB" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm", + "multi-agent-llm" + ], + "arxiv_id": "2607.04240", + "source": "arxiv", + "source_id": "arxiv:2607.04240", + "pdf_url": "https://arxiv.org/pdf/2607.04240", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.03441", + "title": "No Time Like the Present: Agentic Test-Time Training for LLM Agents", + "url": "https://arxiv.org/abs/2607.03441", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Yanbo Wang", + "Jinhua Hao", + "Yuze Shi", + "Kun Yuan", + "Ming Sun" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.03441", + "source": "arxiv", + "source_id": "arxiv:2607.03441", + "pdf_url": "https://arxiv.org/pdf/2607.03441", + "primary_query": "coding-agent" + }, + { + "id": "2607.01874", + "title": "SkillCoach: Self-Evolving Rubrics for Evaluating and Enhancing Agentic Skill-Use", + "url": "https://arxiv.org/abs/2607.01874", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Jiayin Zhu", + "Kelong Mao", + "Yudong Guo", + "Dengbo He", + "Sulong Xu", + "Simiu Gu", + "Yutao Yue" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01874", + "source": "arxiv", + "source_id": "arxiv:2607.01874", + "pdf_url": "https://arxiv.org/pdf/2607.01874", + "primary_query": "llm-agent" + }, + { + "id": "2607.01793", + "title": "Safety Testing LLM Agents at Scale: From Risk Discovery to Evidence-Grounded Verification", + "url": "https://arxiv.org/abs/2607.01793", + "published": "2026-07-02", + "updated": "2026-07-04", + "authors": [ + "Yunhao Feng", + "Ruixiao Lin", + "Ming Wen", + "Qinqin He", + "Yanming Guo", + "Yifan Ding", + "Yutao Wu", + "Jialuo Chen", + "Zhuoer Xu", + "Xiaohu Du", + "Jianan Ma", + "Zixing Chen", + "Xingjun Ma", + "Yunhao Chen", + "Xinhao Deng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01793", + "source": "arxiv", + "source_id": "arxiv:2607.01793", + "pdf_url": "https://arxiv.org/pdf/2607.01793", + "primary_query": "llm-agent" + }, + { + "id": "2607.02703", + "title": "LLMoxie: Exploring Agentic AI for Scientific Software Development", + "url": "https://arxiv.org/abs/2607.02703", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Landung Setiawan", + "Anant Mittal", + "Cordero Core", + "Anshul Tambay", + "Carlos Garcia Jurado Suarez", + "David A. C. Beck", + "Andrew J. Connolly", + "Vani Mandava" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.DC", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2607.02703", + "source": "arxiv", + "source_id": "arxiv:2607.02703", + "pdf_url": "https://arxiv.org/pdf/2607.02703", + "primary_query": "agentic-ai" + }, + { + "id": "2607.00627", + "title": "AGI Maze as a Benchmark Framework for World-Modeling Agents", + "url": "https://arxiv.org/abs/2607.00627", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Alexey Potapov" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00627", + "source": "arxiv", + "source_id": "arxiv:2607.00627", + "pdf_url": "https://arxiv.org/pdf/2607.00627", + "primary_query": "llm-agent" + }, + { + "id": "2607.02606", + "title": "ChainSWE: Benchmarking Coding Agents on Multi-Bug Software Maintenance", + "url": "https://arxiv.org/abs/2607.02606", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Qirui Jin", + "Lingching Tung", + "Kenan Li", + "Qiyang Shi", + "Yushi She", + "Huanzhong Jia", + "Harrison Zhao", + "Kejing Xia", + "Zhenbang Du", + "Yikai Zhang", + "Jiaxin Pei", + "Zhenyu Zhang", + "Zhen Qi", + "Yuyan Duan", + "Wenke Lee", + "Zijian Jin" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02606", + "source": "arxiv", + "source_id": "arxiv:2607.02606", + "pdf_url": "https://arxiv.org/pdf/2607.02606", + "primary_query": "coding-agent" + }, + { + "id": "2606.32025", + "title": "Generative Skill Composition for LLM Agents", + "url": "https://arxiv.org/abs/2606.32025", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Xinyu Zhao", + "Zhen Tan", + "Vaishnav Tadiparthi", + "Nakul Agarwal", + "Kwonjoon Lee", + "Ehsan Moradi Pari", + "Hossein Nourkhiz Mahjoub", + "Tianlong Chen" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.32025", + "source": "arxiv", + "source_id": "arxiv:2606.32025", + "pdf_url": "https://arxiv.org/pdf/2606.32025", + "primary_query": "coding-agent" + }, + { + "id": "2606.31229", + "title": "Agentic-Ideation: Sample Efficient Agentic Trajectories Synthesis for Scientific Ideation Agents", + "url": "https://arxiv.org/abs/2606.31229", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Keyu Zhao", + "Lingyan Kong", + "Fengli Xu", + "Yong Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.31229", + "source": "arxiv", + "source_id": "arxiv:2606.31229", + "pdf_url": "https://arxiv.org/pdf/2606.31229", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31410", + "title": "Xiaomi-GUI-0 Technical Report", + "url": "https://arxiv.org/abs/2606.31410", + "published": "2026-06-30", + "updated": "2026-07-01", + "authors": [ + "Wanxia Cao", + "Chengzhen Duan", + "Pei Fu", + "Pengzhi Gao", + "Niu Lian", + "Fazhan Liu", + "Hui Liu", + "Heng Qu", + "Qinzhuo Wu", + "Zhehao Yu", + "Tongbo Chen", + "Shiqi Cui", + "Anan Du", + "Shukai Jia", + "Yuanfa Li", + "Wei Liu", + "Yike Liu", + "Wenchao Lu", + "Zhenbo Luo", + "Haoyuan Sun", + "Jiatong Sun", + "Cheng Tan", + "Yajie Wang", + "Changqiao Wu", + "Tao Xiong", + "Jiahui Yang", + "Yuxuan Yuan", + "Ruoceng Zhang", + "Shaojie Zhang", + "Jian Zhu", + "Jian Luan", + "Cong Zou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.31410", + "source": "arxiv", + "source_id": "arxiv:2606.31410", + "pdf_url": "https://arxiv.org/pdf/2606.31410", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.29178", + "title": "Selective Memory Retention for Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2606.29178", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Pranath Reddy" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29178", + "source": "arxiv", + "source_id": "arxiv:2606.29178", + "pdf_url": "https://arxiv.org/pdf/2606.29178", + "primary_query": "llm-agent" + }, + { + "id": "2606.28456", + "title": "Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game", + "url": "https://arxiv.org/abs/2606.28456", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Subhendu Bhandary", + "Federico Carucci", + "Christos Charalambous", + "Francesca Dilisante", + "Ksenia Dvorkina", + "Anna Garbo", + "Jiaqi Liang", + "Riccardo Vasellini", + "Francesco Bertolotti" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-safety", + "memory", + "multi-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.28456", + "source": "arxiv", + "source_id": "arxiv:2606.28456", + "pdf_url": "https://arxiv.org/pdf/2606.28456", + "primary_query": "ai-agent" + }, + { + "id": "2606.27406", + "title": "Towards Evaluation of Implicit Software World Models in Coding LLMs", + "url": "https://arxiv.org/abs/2606.27406", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Egor Bogomolov", + "Yaroslav Zharov" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "reasoning", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2606.27406", + "source": "arxiv", + "source_id": "arxiv:2606.27406", + "pdf_url": "https://arxiv.org/pdf/2606.27406", + "primary_query": "ai-agent" + }, + { + "id": "2606.26627", + "title": "Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents", + "url": "https://arxiv.org/abs/2606.26627", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Nada Lahjouji", + "Ashwin Gerard Colaco" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.26627", + "source": "arxiv", + "source_id": "arxiv:2606.26627", + "pdf_url": "https://arxiv.org/pdf/2606.26627", + "primary_query": "agent-memory" + }, + { + "id": "2606.26479", + "title": "Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents", + "url": "https://arxiv.org/abs/2606.26479", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Praneeth Narisetty", + "Shiva Nagendra Babu Kore", + "Uday Kumar Reddy Kattamanchi", + "Jayaram Kumarapu" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.26479", + "source": "arxiv", + "source_id": "arxiv:2606.26479", + "pdf_url": "https://arxiv.org/pdf/2606.26479", + "primary_query": "tool-use" + }, + { + "id": "2606.26793", + "title": "MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG", + "url": "https://arxiv.org/abs/2606.26793", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Inderjeet Singh", + "Andrés Murillo", + "Motoyoshi Sekiya", + "Yuki Unno", + "Junichi Suga" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.26793", + "source": "arxiv", + "source_id": "arxiv:2606.26793", + "pdf_url": "https://arxiv.org/pdf/2606.26793", + "primary_query": "rag-agent" + }, + { + "id": "2606.27472", + "title": "Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents", + "url": "https://arxiv.org/abs/2606.27472", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Vedant Patel" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.27472", + "source": "arxiv", + "source_id": "arxiv:2606.27472", + "pdf_url": "https://arxiv.org/pdf/2606.27472", + "primary_query": "planning-agent" + }, + { + "id": "2606.25622", + "title": "Probabilistic Agents in Deterministic Audits: Evaluating Multi-Agent Systems for Automated Audits Based on the German IT-Grundschutz", + "url": "https://arxiv.org/abs/2606.25622", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Lea Roxanne Muth", + "Marian Margraf" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25622", + "source": "arxiv", + "source_id": "arxiv:2606.25622", + "pdf_url": "https://arxiv.org/pdf/2606.25622", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.25358", + "title": "Agentic Knowledge Tracing: A Multi-Agent LLM Architecture for Stealth Assessment of Financial Literacy in Serious Games", + "url": "https://arxiv.org/abs/2606.25358", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Gabriel Santos", + "Rita Julia", + "Marcelo Nascimento" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25358", + "source": "arxiv", + "source_id": "arxiv:2606.25358", + "pdf_url": "https://arxiv.org/pdf/2606.25358", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.25334", + "title": "Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care", + "url": "https://arxiv.org/abs/2606.25334", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Runwei Guan", + "Yi Zhou", + "Heyi Lin", + "Jinjing Zhu", + "Mingyuan Hou", + "Yang Yang", + "Fang Yuan", + "Xiaohong Lin", + "Shaofeng Liang", + "Xuming Hu", + "Tao Li", + "Tianbin Zhao", + "Yutao Yue", + "Zhiyuan Wang", + "Hui Xiong" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "rag" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25334", + "source": "arxiv", + "source_id": "arxiv:2606.25334", + "pdf_url": "https://arxiv.org/pdf/2606.25334", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22673", + "title": "AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent", + "url": "https://arxiv.org/abs/2606.22673", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Weidi Luo", + "Qiming Zhang", + "Yihao Quan", + "Mingyu Jin", + "Jie Cai", + "Chaowei Xiao", + "Jingcheng Niu", + "Zhen Xiang" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "coding-agent" + ], + "arxiv_id": "2606.22673", + "source": "arxiv", + "source_id": "arxiv:2606.22673", + "pdf_url": "https://arxiv.org/pdf/2606.22673", + "primary_query": "agent-safety" + }, + { + "id": "2606.21710", + "title": "PrivacyAlign: Contextual Privacy Alignment for LLM Agents", + "url": "https://arxiv.org/abs/2606.21710", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Manveer Singh Tamber", + "Abhay Puri", + "Marc-Etienne Brunet", + "Perouz Taslakian", + "Jimmy Lin", + "Spandana Gella" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.21710", + "source": "arxiv", + "source_id": "arxiv:2606.21710", + "pdf_url": "https://arxiv.org/pdf/2606.21710", + "primary_query": "ai-agent" + }, + { + "id": "2606.21013", + "title": "Agentic Time Machine as an Infrastructure for Future-Event Forecasting", + "url": "https://arxiv.org/abs/2606.21013", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Jingyi Chai", + "Bingyang Zheng", + "Xiangrui Liu", + "Hao Lu", + "Zihang Zhou", + "Tianchen Wang", + "Kemeng Zhang", + "Siheng Chen" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.21013", + "source": "arxiv", + "source_id": "arxiv:2606.21013", + "pdf_url": "https://arxiv.org/pdf/2606.21013", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.21123", + "title": "A Multi-Agent Audit Framework for High-Stakes Reasoning: Evaluation and Interpretability in Clinical Mental Health Screening", + "url": "https://arxiv.org/abs/2606.21123", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Jingchen Ye", + "Yanpei Yu", + "Luyao Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.21123", + "source": "arxiv", + "source_id": "arxiv:2606.21123", + "pdf_url": "https://arxiv.org/pdf/2606.21123", + "primary_query": "rag-agent" + }, + { + "id": "2606.19899", + "title": "Measuring Biological Capabilities and Risks of AI Agents", + "url": "https://arxiv.org/abs/2606.19899", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Patricia Paskov", + "Jeffrey Lee", + "Kyle Brady", + "Alyssa Worland" + ], + "categories": [ + "cs.CY", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "agentic-ai", + "ai-agent" + ], + "arxiv_id": "2606.19899", + "source": "arxiv", + "source_id": "arxiv:2606.19899", + "pdf_url": "https://arxiv.org/pdf/2606.19899", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.20243", + "title": "Phoenix: Safe GitHub Issue Resolution via Multi-Agent LLMs", + "url": "https://arxiv.org/abs/2606.20243", + "published": "2026-06-18", + "updated": "2026-06-22", + "authors": [ + "Kipngeno Koech", + "Muhammad Adam", + "Baimam Boukar Jean Jacques", + "Joao Barros" + ], + "categories": [ + "cs.SE", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "multi-agent", + "planning", + "rag" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.20243", + "source": "arxiv", + "source_id": "arxiv:2606.20243", + "pdf_url": "https://arxiv.org/pdf/2606.20243", + "primary_query": "coding-agent" + }, + { + "id": "2606.18950", + "title": "RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models", + "url": "https://arxiv.org/abs/2606.18950", + "published": "2026-06-17", + "updated": "2026-06-18", + "authors": [ + "San Kim", + "Daechul Ahn", + "Reokyoung Kim", + "Hyeonbeom Choi", + "Seungyeon Jwa", + "Jonghyun Choi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.18950", + "source": "arxiv", + "source_id": "arxiv:2606.18950", + "pdf_url": "https://arxiv.org/pdf/2606.18950", + "primary_query": "agent-memory" + }, + { + "id": "2606.16659", + "title": "FraudSMSWalker: Benchmarking Agentic Large Language Models for SMS-to-Webpage Fraud Detection", + "url": "https://arxiv.org/abs/2606.16659", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Y. H. Zhou", + "Z. M. Ma", + "Y. J. Zhou", + "Y. T. Li", + "H. X. Xiang", + "Y. M. Cheng", + "T. L. Chen", + "K. J. Zhang", + "Z. H. Nan", + "J. H. Ni", + "Z. Wu", + "Q. Y. Pan", + "S. Zhang", + "S. Cheng", + "M. Y. Luo" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.16659", + "source": "arxiv", + "source_id": "arxiv:2606.16659", + "pdf_url": "https://arxiv.org/pdf/2606.16659", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.17041", + "title": "Benchmarking LLM Agents on Meta-Analysis Articles from Nature Portfolio", + "url": "https://arxiv.org/abs/2606.17041", + "published": "2026-06-15", + "updated": "2026-07-01", + "authors": [ + "Anzhe Xie", + "Weihang Su", + "Yujia Zhou", + "Yiqun Liu", + "Qingyao Ai" + ], + "categories": [ + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.17041", + "source": "arxiv", + "source_id": "arxiv:2606.17041", + "pdf_url": "https://arxiv.org/pdf/2606.17041", + "primary_query": "rag-agent" + }, + { + "id": "2606.16576", + "title": "Can LLM Agents Infer World Models? Evidence from Agentic Automata Learning", + "url": "https://arxiv.org/abs/2606.16576", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Reef Menaged", + "Gili Lior", + "Shauli Ravfogel", + "Roee Aharoni", + "Gabriel Stanovsky" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.16576", + "source": "arxiv", + "source_id": "arxiv:2606.16576", + "pdf_url": "https://arxiv.org/pdf/2606.16576", + "primary_query": "planning-agent" + }, + { + "id": "2606.12780", + "title": "ProPlay: Procedural World Models for Self-Evolving LLM Agents", + "url": "https://arxiv.org/abs/2606.12780", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Yijun Ma", + "Zehong Wang", + "Yiyang Li", + "Ziming Li", + "Xiaoguang Guo", + "Weixiang Sun", + "Chuxu Zhang", + "Yanfang Ye" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "tool-use", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.12780", + "source": "arxiv", + "source_id": "arxiv:2606.12780", + "pdf_url": "https://arxiv.org/pdf/2606.12780", + "primary_query": "planning-agent" + }, + { + "id": "2606.12320", + "title": "A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents", + "url": "https://arxiv.org/abs/2606.12320", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Krti Tallam" + ], + "categories": [ + "cs.AI", + "cs.CC", + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.12320", + "source": "arxiv", + "source_id": "arxiv:2606.12320", + "pdf_url": "https://arxiv.org/pdf/2606.12320", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.11680", + "title": "Organize then Retrieve: Hierarchical Memory Navigation for Efficient Agents", + "url": "https://arxiv.org/abs/2606.11680", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Hao-Lun Hsu", + "Nikki Lijing Kuang", + "Boyi Liu", + "Zhewei Yao", + "Yuxiong He" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.11680", + "source": "arxiv", + "source_id": "arxiv:2606.11680", + "pdf_url": "https://arxiv.org/pdf/2606.11680", + "primary_query": "agent-memory" + }, + { + "id": "2606.12657", + "title": "TrajGenAgent: A Hierarchical LLM Agent for Human Mobility Trajectory Generation", + "url": "https://arxiv.org/abs/2606.12657", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Siyu Li", + "Toan Tran", + "Lingyi Zhao", + "Khurram Shafique", + "Li Xiong" + ], + "categories": [ + "cs.AI", + "cs.DB", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "workflow-agent", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.12657", + "source": "arxiv", + "source_id": "arxiv:2606.12657", + "pdf_url": "https://arxiv.org/pdf/2606.12657", + "primary_query": "planning-agent" + }, + { + "id": "2606.10677", + "title": "Infini Memory: Maintainable Topic Documents for Long-Term LLM Agent Memory", + "url": "https://arxiv.org/abs/2606.10677", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Suozhao Ji", + "Baodong Wu", + "Zehao Wang", + "Lei Xia", + "Qingping Li", + "Ruisong Wang", + "Wenbo Ding", + "Zhenhua Zhu", + "Boxun Li", + "Guohao Dai", + "Yu Wang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.10677", + "source": "arxiv", + "source_id": "arxiv:2606.10677", + "pdf_url": "https://arxiv.org/pdf/2606.10677", + "primary_query": "agent-memory" + }, + { + "id": "2606.11354", + "title": "A Zero-Shot Multi-Agent Framework for Human-Building Interaction via Programmatic Reasoning", + "url": "https://arxiv.org/abs/2606.11354", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yuqi Wang", + "Gulai Shen", + "Ali Mehmani" + ], + "categories": [ + "cs.ET" + ], + "topics": [ + "agent-safety", + "coding-agent", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.11354", + "source": "arxiv", + "source_id": "arxiv:2606.11354", + "pdf_url": "https://arxiv.org/pdf/2606.11354", + "primary_query": "rag-agent" + }, + { + "id": "2606.05558", + "title": "Autoregressive Diffusion World Models for Off-Policy Evaluation of LLM Agents", + "url": "https://arxiv.org/abs/2606.05558", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Kaixuan Liu", + "Guojun Xiong", + "Weinan Zhang", + "Shengpu Tang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.05558", + "source": "arxiv", + "source_id": "arxiv:2606.05558", + "pdf_url": "https://arxiv.org/pdf/2606.05558", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.04555", + "title": "Temporal Order Matters for Agentic Memory: Segment Trees for Long-Horizon Agents", + "url": "https://arxiv.org/abs/2606.04555", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Yifan Simon Liu", + "Liam Gallagher", + "Faeze Moradi Kalarde", + "Jiazhou Liang", + "Armin Toroghi", + "Scott Sanner" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04555", + "source": "arxiv", + "source_id": "arxiv:2606.04555", + "pdf_url": "https://arxiv.org/pdf/2606.04555", + "primary_query": "agent-memory" + }, + { + "id": "2606.03895", + "title": "Agent libOS: A Runtime Substrate for Capability-Controlled Self-Evolving LLM Agents", + "url": "https://arxiv.org/abs/2606.03895", + "published": "2026-06-02", + "updated": "2026-06-29", + "authors": [ + "Yingqi Zhang" + ], + "categories": [ + "cs.OS", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.03895", + "source": "arxiv", + "source_id": "arxiv:2606.03895", + "pdf_url": "https://arxiv.org/pdf/2606.03895", + "primary_query": "planning-agent" + }, + { + "id": "2606.02302", + "title": "SeClaw: Spec-Driven Security Task Synthesis for Evaluating Autonomous Agents", + "url": "https://arxiv.org/abs/2606.02302", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Hao Cheng", + "Changtao Miao", + "Tianle Song", + "Yin Wu", + "He Liu", + "Erjia Xiao", + "Junchi Chen", + "Xiaoyu Shi", + "Yichi Wang", + "Jing Yang", + "Taowen Wang", + "Jinhao Duan", + "Mengshu Sun", + "Peiyan Dong", + "Xuan Shen", + "Yang Cao", + "Renjing Xu", + "Kaidi Xu", + "Jindong Gu", + "Bo Zhang", + "Jize Zhang", + "Chenhao Lin", + "Philip Torr", + "Chao Shen" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.02302", + "source": "arxiv", + "source_id": "arxiv:2606.02302", + "pdf_url": "https://arxiv.org/pdf/2606.02302", + "primary_query": "agent-safety" + }, + { + "id": "2605.30711", + "title": "SAGE: A Novelty Gate for Efficient Memory Evolution in Agentic LLMs", + "url": "https://arxiv.org/abs/2605.30711", + "published": "2026-05-29", + "updated": "2026-06-18", + "authors": [ + "Sijia Wang", + "Dhanajit Brahma", + "Ricardo Henao" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG", + "stat.ML" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.30711", + "source": "arxiv", + "source_id": "arxiv:2605.30711", + "pdf_url": "https://arxiv.org/pdf/2605.30711", + "primary_query": "agent-memory" + }, + { + "id": "2605.29341", + "title": "WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction", + "url": "https://arxiv.org/abs/2605.29341", + "published": "2026-05-28", + "updated": "2026-06-01", + "authors": [ + "Chengzhi Liu", + "Yuzhe Yang", + "Sophia Xiao Pu", + "Yepeng Liu", + "Lin Long", + "Yichen Guo", + "Nuo Chen", + "Zhaotian Weng", + "Elena Kochkina", + "Simerjot Kaur", + "Charese Smiley", + "Xiaomo Liu", + "James Zou", + "Sheng Liu", + "Yuheng Bu", + "Songyou Peng", + "Xin Eric Wang" + ], + "categories": [ + "cs.CV", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2605.29341", + "source": "arxiv", + "source_id": "arxiv:2605.29341", + "pdf_url": "https://arxiv.org/pdf/2605.29341", + "primary_query": "agent-memory" + }, + { + "id": "2605.27690", + "title": "TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling", + "url": "https://arxiv.org/abs/2605.27690", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Jiaqian Li", + "Yanshu Li", + "Boxuan Zhang", + "Ruixiang Tang", + "Kuan-Hao Huang" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.27690", + "source": "arxiv", + "source_id": "arxiv:2605.27690", + "pdf_url": "https://arxiv.org/pdf/2605.27690", + "primary_query": "agent-safety" + }, + { + "id": "2605.25141", + "title": "LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support", + "url": "https://arxiv.org/abs/2605.25141", + "published": "2026-05-24", + "updated": "2026-05-24", + "authors": [ + "Pavan Manjunath", + "Thomas Pruefer" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.25141", + "source": "arxiv", + "source_id": "arxiv:2605.25141", + "pdf_url": "https://arxiv.org/pdf/2605.25141", + "primary_query": "planning-agent" + }, + { + "id": "2605.19952", + "title": "Rethinking How to Remember: Beyond Atomic Facts in Lifelong LLM Agent Memory", + "url": "https://arxiv.org/abs/2605.19952", + "published": "2026-05-19", + "updated": "2026-05-19", + "authors": [ + "Jingwei Sun", + "Jianing Zhu", + "Jiangchao Yao", + "Tongliang Liu", + "Bo Han" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.19952", + "source": "arxiv", + "source_id": "arxiv:2605.19952", + "pdf_url": "https://arxiv.org/pdf/2605.19952", + "primary_query": "agent-memory" + }, + { + "id": "2605.17625", + "title": "Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents", + "url": "https://arxiv.org/abs/2605.17625", + "published": "2026-05-17", + "updated": "2026-05-17", + "authors": [ + "Nikola Milosevic" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.17625", + "source": "arxiv", + "source_id": "arxiv:2605.17625", + "pdf_url": "https://arxiv.org/pdf/2605.17625", + "primary_query": "agent-memory" + }, + { + "id": "2605.16821", + "title": "Multi-Paradigm Agent Interaction in Practice:A Systematic Analysis of Generator-Evaluator, ReAct Loop,and Adversarial Evaluation in the buddyMe Framework", + "url": "https://arxiv.org/abs/2605.16821", + "published": "2026-05-16", + "updated": "2026-05-16", + "authors": [ + "Xiaohua Wang", + "Chao Han", + "Kai Yu", + "XiaoLiang Xu", + "Liang Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.16821", + "source": "arxiv", + "source_id": "arxiv:2605.16821", + "pdf_url": "https://arxiv.org/pdf/2605.16821", + "primary_query": "planning-agent" + }, + { + "id": "2605.16481", + "title": "Visual Agentic Memory: Enabling Online Long Video Understanding via Online Indexing, Hierarchical Memory, and Agentic Retrieval", + "url": "https://arxiv.org/abs/2605.16481", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Aiden Yiliu Li", + "Nels Numan", + "Anthony Steed" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.16481", + "source": "arxiv", + "source_id": "arxiv:2605.16481", + "pdf_url": "https://arxiv.org/pdf/2605.16481", + "primary_query": "agent-memory" + }, + { + "id": "2605.12061", + "title": "SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory", + "url": "https://arxiv.org/abs/2605.12061", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Juntong Wang", + "Haoyue Zhao", + "guanghui Pan", + "Xiyuan Wang", + "Yanbo Wang", + "Qiyan Deng", + "Muhan Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.12061", + "source": "arxiv", + "source_id": "arxiv:2605.12061", + "pdf_url": "https://arxiv.org/pdf/2605.12061", + "primary_query": "language-agent" + }, + { + "id": "2605.06890", + "title": "Beyond the Black Box: Interpretability of Agentic AI Tool Use", + "url": "https://arxiv.org/abs/2605.06890", + "published": "2026-05-07", + "updated": "2026-07-05", + "authors": [ + "Hariom Tatsat", + "Ariye Shater" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.06890", + "source": "arxiv", + "source_id": "arxiv:2605.06890", + "pdf_url": "https://arxiv.org/pdf/2605.06890", + "primary_query": "function-calling" + }, + { + "id": "2605.05716", + "title": "More Is Not Always Better: Cross-Component Interference in LLM Agent Scaffolding", + "url": "https://arxiv.org/abs/2605.05716", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Ming Liu" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.05716", + "source": "arxiv", + "source_id": "arxiv:2605.05716", + "pdf_url": "https://arxiv.org/pdf/2605.05716", + "primary_query": "planning-agent" + }, + { + "id": "2605.06716", + "title": "From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms", + "url": "https://arxiv.org/abs/2605.06716", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Jinghao Luo", + "Yuchen Tian", + "Chuxue Cao", + "Ziyang Luo", + "Hongzhan Lin", + "Kaixin Li", + "Chuyi Kong", + "Ruichao Yang", + "Jing Ma" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.06716", + "source": "arxiv", + "source_id": "arxiv:2605.06716", + "pdf_url": "https://arxiv.org/pdf/2605.06716", + "primary_query": "planning-agent" + }, + { + "id": "2604.25318", + "title": "Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation", + "url": "https://arxiv.org/abs/2604.25318", + "published": "2026-04-28", + "updated": "2026-04-28", + "authors": [ + "Lanshan He", + "Haozhou Pang", + "Qi Gan", + "Xin Shen", + "Ziwei Zhang", + "Yibo Liu", + "Gang Fang", + "Bo Liu", + "Kai Sheng", + "Shengfeng Zeng", + "Chaofan Li", + "Zhen Hui", + "Keer Zhou", + "Lan Zhou", + "Shujun Dai" + ], + "categories": [ + "cs.GR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2604.25318", + "source": "arxiv", + "source_id": "arxiv:2604.25318", + "pdf_url": "https://arxiv.org/pdf/2604.25318", + "primary_query": "function-calling" + }, + { + "id": "2604.25135", + "title": "FAMA: Failure-Aware Meta-Agentic Framework for Open-Source LLMs in Interactive Tool Use Environments", + "url": "https://arxiv.org/abs/2604.25135", + "published": "2026-04-28", + "updated": "2026-04-28", + "authors": [ + "Amir Saeidi", + "Venkatesh Mishra", + "Souradeep Mukhopadhyay", + "Gaowen Liu", + "Ali Payani", + "Jayanth Srinivasa", + "Chitta Baral" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.25135", + "source": "arxiv", + "source_id": "arxiv:2604.25135", + "pdf_url": "https://arxiv.org/pdf/2604.25135", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.23459", + "title": "Architecture Matters for Multi-Agent Security", + "url": "https://arxiv.org/abs/2604.23459", + "published": "2026-04-25", + "updated": "2026-04-25", + "authors": [ + "Ben Hagag", + "William L. Anderson", + "Christian Schroeder de Witt", + "Sarah Scheffler" + ], + "categories": [ + "cs.MA", + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "planning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.23459", + "source": "arxiv", + "source_id": "arxiv:2604.23459", + "pdf_url": "https://arxiv.org/pdf/2604.23459", + "primary_query": "agent-safety" + }, + { + "id": "2604.23374", + "title": "Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents", + "url": "https://arxiv.org/abs/2604.23374", + "published": "2026-04-25", + "updated": "2026-04-25", + "authors": [ + "Yuandao Cai", + "Wensheng Tang", + "Cheng Wen", + "Shengchao Qin" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.23374", + "source": "arxiv", + "source_id": "arxiv:2604.23374", + "pdf_url": "https://arxiv.org/pdf/2604.23374", + "primary_query": "agent-safety" + }, + { + "id": "2604.22879", + "title": "Beyond Single-Agent Alignment: Preventing Context-Fragmented Violations in Multi-Agent Systems", + "url": "https://arxiv.org/abs/2604.22879", + "published": "2026-04-24", + "updated": "2026-04-24", + "authors": [ + "Jie Wu", + "Ming Gong" + ], + "categories": [ + "cs.MA", + "cs.AI", + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.22879", + "source": "arxiv", + "source_id": "arxiv:2604.22879", + "pdf_url": "https://arxiv.org/pdf/2604.22879", + "primary_query": "agent-safety" + }, + { + "id": "2604.18847", + "title": "Human-Guided Harm Recovery for Computer Use Agents", + "url": "https://arxiv.org/abs/2604.18847", + "published": "2026-04-20", + "updated": "2026-05-28", + "authors": [ + "Christy Li", + "Sky CH-Wang", + "Andi Peng", + "Andreea Bobu" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.18847", + "source": "arxiv", + "source_id": "arxiv:2604.18847", + "pdf_url": "https://arxiv.org/pdf/2604.18847", + "primary_query": "agent-safety" + }, + { + "id": "2603.18245", + "title": "Who Tests the Testers? Systematic Enumeration and Coverage Audit of LLM Agent Tool Call Safety", + "url": "https://arxiv.org/abs/2603.18245", + "published": "2026-03-18", + "updated": "2026-03-18", + "authors": [ + "Xuan Chen", + "Lu Yan", + "Ruqi Zhang", + "Xiangyu Zhang" + ], + "categories": [ + "cs.SE", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.18245", + "source": "arxiv", + "source_id": "arxiv:2603.18245", + "pdf_url": "https://arxiv.org/pdf/2603.18245", + "primary_query": "agent-safety" + }, + { + "id": "2603.07980", + "title": "\\$OneMillion-Bench: How Far are Language Agents from Human Experts?", + "url": "https://arxiv.org/abs/2603.07980", + "published": "2026-03-09", + "updated": "2026-03-09", + "authors": [ + "Qianyu Yang", + "Yang Liu", + "Jiaqi Li", + "Jun Bai", + "Hao Chen", + "Kaiyuan Chen", + "Tiliang Duan", + "Jiayun Dong", + "Xiaobo Hu", + "Zixia Jia", + "Yang Liu", + "Tao Peng", + "Yixin Ren", + "Ran Tian", + "Zaiyuan Wang", + "Yanglihong Xiao", + "Gang Yao", + "Lingyue Yin", + "Ge Zhang", + "Chun Zhang", + "Jianpeng Jiao", + "Zilong Zheng", + "Yuan Gong" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.07980", + "source": "arxiv", + "source_id": "arxiv:2603.07980", + "pdf_url": "https://arxiv.org/pdf/2603.07980", + "primary_query": "language-agent" + }, + { + "id": "2603.09002", + "title": "Security Considerations for Multi-agent Systems", + "url": "https://arxiv.org/abs/2603.09002", + "published": "2026-03-09", + "updated": "2026-04-26", + "authors": [ + "Tam Nguyen", + "Moses Ndebugre", + "Dheeraj Arremsetty" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.09002", + "source": "arxiv", + "source_id": "arxiv:2603.09002", + "pdf_url": "https://arxiv.org/pdf/2603.09002", + "primary_query": "agent-safety" + }, + { + "id": "2602.07962", + "title": "LOCA-bench: Benchmarking Language Agents Under Controllable and Extreme Context Growth", + "url": "https://arxiv.org/abs/2602.07962", + "published": "2026-02-08", + "updated": "2026-02-08", + "authors": [ + "Weihao Zeng", + "Yuzhen Huang", + "Junxian He" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.07962", + "source": "arxiv", + "source_id": "arxiv:2602.07962", + "pdf_url": "https://arxiv.org/pdf/2602.07962", + "primary_query": "language-agent" + }, + { + "id": "2602.05302", + "title": "PieArena: Ranking and Profiling Language Agents in Realistic Negotiation Scenarios", + "url": "https://arxiv.org/abs/2602.05302", + "published": "2026-02-05", + "updated": "2026-06-01", + "authors": [ + "Chris Zhu", + "Sasha Cui", + "Will Sanok Dufallo", + "Runzhi Jin", + "Zhen Xu", + "Linjun Zhang", + "Daylian Cain" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.05302", + "source": "arxiv", + "source_id": "arxiv:2602.05302", + "pdf_url": "https://arxiv.org/pdf/2602.05302", + "primary_query": "language-agent" + }, + { + "id": "2601.06007", + "title": "Don't Break the Cache: An Evaluation of Prompt Caching for Long-Horizon Agentic Tasks", + "url": "https://arxiv.org/abs/2601.06007", + "published": "2026-01-09", + "updated": "2026-01-31", + "authors": [ + "Elias Lumer", + "Faheem Nizar", + "Akshaya Jangiti", + "Kevin Frank", + "Anmol Gulati", + "Mandar Phadate", + "Vamse Kumar Subbiah" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.06007", + "source": "arxiv", + "source_id": "arxiv:2601.06007", + "pdf_url": "https://arxiv.org/pdf/2601.06007", + "primary_query": "function-calling" + }, + { + "id": "2511.04847", + "title": "Test-Time Adaptation for LLM Agents via Environment Interaction", + "url": "https://arxiv.org/abs/2511.04847", + "published": "2025-11-06", + "updated": "2026-02-22", + "authors": [ + "Arthur Chen", + "Zuxin Liu", + "Jianguo Zhang", + "Akshara Prabhakar", + "Zhiwei Liu", + "Shelby Heinecke", + "Silvio Savarese", + "Victor Zhong", + "Caiming Xiong" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2511.04847", + "source": "arxiv", + "source_id": "arxiv:2511.04847", + "pdf_url": "https://arxiv.org/pdf/2511.04847", + "primary_query": "function-calling" + }, + { + "id": "2509.10769", + "title": "AgentArch: A Comprehensive Benchmark to Evaluate Agent Architectures in Enterprise", + "url": "https://arxiv.org/abs/2509.10769", + "published": "2025-09-13", + "updated": "2026-01-06", + "authors": [ + "Tara Bogavelli", + "Roshnee Sharma", + "Hari Subramani" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.10769", + "source": "arxiv", + "source_id": "arxiv:2509.10769", + "pdf_url": "https://arxiv.org/pdf/2509.10769", + "primary_query": "function-calling" + }, + { + "id": "2607.06273", + "title": "AgentTether: Graph-Guided Diagnosis and Runtime Intervention for Reliable LLM Agent Operation", + "url": "https://arxiv.org/abs/2607.06273", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Chenyu Zhao", + "Shenglin Zhang", + "Wenwei Gu", + "Yongqian Sun", + "Dan Pei", + "Chetan Bansal", + "Saravan Rajmohan", + "Minghua Ma" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.06273", + "source": "arxiv", + "source_id": "arxiv:2607.06273", + "pdf_url": "https://arxiv.org/pdf/2607.06273", + "primary_query": "llm-agent" + }, + { + "id": "2607.06195", + "title": "LogicHunter: Testing LLM Agent Frameworks with an Agentic Oracle", + "url": "https://arxiv.org/abs/2607.06195", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Minghui Long", + "Yanjie Zhao", + "Haoyu Wang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.06195", + "source": "arxiv", + "source_id": "arxiv:2607.06195", + "pdf_url": "https://arxiv.org/pdf/2607.06195", + "primary_query": "llm-agent" + }, + { + "id": "2607.06080", + "title": "From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations", + "url": "https://arxiv.org/abs/2607.06080", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Shiyi Ling", + "Zhi Zheng", + "Hui Zheng", + "Wenjun Xue", + "Feng Ye", + "Tong Xu" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.SI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.06080", + "source": "arxiv", + "source_id": "arxiv:2607.06080", + "pdf_url": "https://arxiv.org/pdf/2607.06080", + "primary_query": "llm-agent" + }, + { + "id": "2607.05297", + "title": "MetaSkill-Evolve: Recursive Self-Improvement of LLM Agents via Two-Timescale Meta-Skill Evolution", + "url": "https://arxiv.org/abs/2607.05297", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Zefeng Wang", + "Minxi Yan", + "Jinhe Bi", + "Sikuan Yan", + "Volker Tresp", + "Yunpu Ma" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent" + ], + "arxiv_id": "2607.05297", + "source": "arxiv", + "source_id": "arxiv:2607.05297", + "pdf_url": "https://arxiv.org/pdf/2607.05297", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.04528", + "title": "Measuring Harness-Induced Belief Divergence in Multi-Step LLM Agents", + "url": "https://arxiv.org/abs/2607.04528", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Haiwen Yi", + "Xinyuan Song" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent" + ], + "arxiv_id": "2607.04528", + "source": "arxiv", + "source_id": "arxiv:2607.04528", + "pdf_url": "https://arxiv.org/pdf/2607.04528", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.04293", + "title": "CausalGame: Benchmarking Causal Thinking of LLM Agents in Games", + "url": "https://arxiv.org/abs/2607.04293", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Zhenhao Chen", + "Yongqiang Chen", + "Chenxi Liu", + "Junchi Yu", + "Xiangchen Song", + "Zijian Li", + "Jialin Li", + "Philip Torr", + "Bo Han", + "Kun Zhang" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG", + "stat.ML" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04293", + "source": "arxiv", + "source_id": "arxiv:2607.04293", + "pdf_url": "https://arxiv.org/pdf/2607.04293", + "primary_query": "llm-agent" + }, + { + "id": "2607.04426", + "title": "ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI", + "url": "https://arxiv.org/abs/2607.04426", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "ACE-Brain Team", + ":", + "Ziyang Gong", + "Haoming Gu", + "Zehang Luo", + "Tianyi Zhang", + "Tao Tao", + "Yixiao Chi", + "Zhe Liu", + "Lingsi Zhu", + "Jingyuan Liu", + "Anke Tang", + "Songze Li", + "Yilun Kong", + "Ningjing Liu", + "Tianyu Zhu", + "Yunpeng Qing", + "Shuang Luo", + "Xiang Liu", + "Shi Fu", + "Dawei Nie", + "Sixiang Liu", + "Zhexi Wen", + "Feng Pan", + "Xiaofeng Wang", + "Zhi Hou", + "Chunxiao Liu", + "Xue Yang", + "Junchi Yan", + "Hengshuang Zhao", + "Dacheng Tao", + "Xiaogang Wang" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04426", + "source": "arxiv", + "source_id": "arxiv:2607.04426", + "pdf_url": "https://arxiv.org/pdf/2607.04426", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04162", + "title": "ACE: Agentic Control for Embodied Manipulation via Zero-shot Workflow Reasoning", + "url": "https://arxiv.org/abs/2607.04162", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Iok Tong Lei", + "QianZhi Li", + "Ying Jie Yap", + "Yujie Zhang", + "Rui Zhong", + "Haichao Gui", + "Xiaolong Liu", + "Zhidong Deng" + ], + "categories": [ + "cs.RO", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04162", + "source": "arxiv", + "source_id": "arxiv:2607.04162", + "pdf_url": "https://arxiv.org/pdf/2607.04162", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03333", + "title": "SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference", + "url": "https://arxiv.org/abs/2607.03333", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Huajun Bai", + "Weiwei Lv", + "Huichuan Zheng", + "Youyou Lu", + "Jiwu Shu" + ], + "categories": [ + "cs.DC", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.03333", + "source": "arxiv", + "source_id": "arxiv:2607.03333", + "pdf_url": "https://arxiv.org/pdf/2607.03333", + "primary_query": "llm-agent" + }, + { + "id": "2607.02857", + "title": "MOSAIC: Knowledge-Guided CLI Command Composition Attack in LLM Coding Agents", + "url": "https://arxiv.org/abs/2607.02857", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Jiangrong Wu", + "Huaijin Wang", + "Yihao Zhang", + "Yuhong Nan", + "Shuai Wang" + ], + "categories": [ + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2607.02857", + "source": "arxiv", + "source_id": "arxiv:2607.02857", + "pdf_url": "https://arxiv.org/pdf/2607.02857", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.01935", + "title": "A-TMA: Decoupling State-Aware Memory Failures in Long-Term Agent Memory", + "url": "https://arxiv.org/abs/2607.01935", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Zitong Shi", + "Yixuan Tang", + "Anthony Kum Hoe Tung" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "llm-agent" + ], + "arxiv_id": "2607.01935", + "source": "arxiv", + "source_id": "arxiv:2607.01935", + "pdf_url": "https://arxiv.org/pdf/2607.01935", + "primary_query": "agent-memory" + }, + { + "id": "2607.01640", + "title": "AgentFlow: Building Agent Dependency Graphs for Static Analysis of Agent Programs", + "url": "https://arxiv.org/abs/2607.01640", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Shenao Wang", + "Xinyi Hou", + "Yanjie Zhao", + "Xiao Cheng", + "Haoyu Wang" + ], + "categories": [ + "cs.SE", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.01640", + "source": "arxiv", + "source_id": "arxiv:2607.01640", + "pdf_url": "https://arxiv.org/pdf/2607.01640", + "primary_query": "llm-agent" + }, + { + "id": "2607.01668", + "title": "VeriChat: An Agentic Conversational AI Assistant for Hardware Security Verification", + "url": "https://arxiv.org/abs/2607.01668", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Dipayan Saha", + "Khan Thamid Hasan", + "Shams Tarek", + "Sujan Kumar Saha", + "Mark Tehranipoor", + "Farimah Farahmandi" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.01668", + "source": "arxiv", + "source_id": "arxiv:2607.01668", + "pdf_url": "https://arxiv.org/pdf/2607.01668", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02294", + "title": "Coding Agents Are Guessing: Measuring Action-Boundary Violations in Underspecified DevOps Instructions", + "url": "https://arxiv.org/abs/2607.02294", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Zimo Ji", + "Zekai Zhang", + "Congying Xu", + "Zongjie Li", + "Yudong Gao", + "Shuai Wang", + "Shing-Chi Cheung" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2607.02294", + "source": "arxiv", + "source_id": "arxiv:2607.02294", + "pdf_url": "https://arxiv.org/pdf/2607.02294", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.01916", + "title": "ContextSniper: AntTrail's Token-Efficient Code Memory for Repository-Level Program Repair", + "url": "https://arxiv.org/abs/2607.01916", + "published": "2026-07-02", + "updated": "2026-07-06", + "authors": [ + "Chiwang Luk", + "Matin Mohammad Najafi", + "Zhifeng Jia", + "Wei Yang", + "Xiuchang Li", + "Jinwei Zhu", + "Yang Ren", + "Lei Chen", + "Gao Cong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "coding-agent", + "rag-agent" + ], + "arxiv_id": "2607.01916", + "source": "arxiv", + "source_id": "arxiv:2607.01916", + "pdf_url": "https://arxiv.org/pdf/2607.01916", + "primary_query": "agent-memory" + }, + { + "id": "2607.01929", + "title": "Beyond Textual Repository Exploration: Dual-Modal Structural Reasoning for Agentic Issue Resolution", + "url": "https://arxiv.org/abs/2607.01929", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Jiayi Zhang", + "Kai Huang", + "Yang Liu", + "Chunyang Chen" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "function-calling" + ], + "arxiv_id": "2607.01929", + "source": "arxiv", + "source_id": "arxiv:2607.01929", + "pdf_url": "https://arxiv.org/pdf/2607.01929", + "primary_query": "coding-agent" + }, + { + "id": "2607.01071", + "title": "MemSyco-Bench: Benchmarking Sycophancy in Agent Memory", + "url": "https://arxiv.org/abs/2607.01071", + "published": "2026-07-01", + "updated": "2026-07-02", + "authors": [ + "Zhishang Xiang", + "Zerui Chen", + "Yunbo Tang", + "Zhimin Wei", + "Ruqin Ning", + "Yujie Lin", + "Qinggang Zhang", + "Jinsong Su" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2607.01071", + "source": "arxiv", + "source_id": "arxiv:2607.01071", + "pdf_url": "https://arxiv.org/pdf/2607.01071", + "primary_query": "agent-memory" + }, + { + "id": "2607.00939", + "title": "Leveraging LLM-Based Agentic Systems to Generate Quantum Applications for Test Optimization", + "url": "https://arxiv.org/abs/2607.00939", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Ming Tao", + "Yuechen Li", + "Tao Yue", + "Man Zhang", + "Aitor Arrieta Marcos" + ], + "categories": [ + "cs.SE", + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00939", + "source": "arxiv", + "source_id": "arxiv:2607.00939", + "pdf_url": "https://arxiv.org/pdf/2607.00939", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00334", + "title": "Managed Autonomy at Runtime: Gear-Based Safety and Governance for Single- and Multi-Agent Cyber-Physical Systems", + "url": "https://arxiv.org/abs/2607.00334", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Srini Ramaswamy", + "Wang Miaosheng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "multi-agent", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "multi-agent-llm" + ], + "arxiv_id": "2607.00334", + "source": "arxiv", + "source_id": "arxiv:2607.00334", + "pdf_url": "https://arxiv.org/pdf/2607.00334", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.05428", + "title": "CHARLIE: An On-Premise Multi-Agent Retrieval-Augmented Generation System for Evidential Reasoning in Forensic Science", + "url": "https://arxiv.org/abs/2607.05428", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Leandro D. Carneiro", + "Andre L. S. Meirelles", + "Juliano de A. Gomes", + "Rafael C. A. Cabral" + ], + "categories": [ + "cs.DL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.05428", + "source": "arxiv", + "source_id": "arxiv:2607.05428", + "pdf_url": "https://arxiv.org/pdf/2607.05428", + "primary_query": "rag-agent" + }, + { + "id": "2606.31693", + "title": "ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping", + "url": "https://arxiv.org/abs/2606.31693", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Jiacheng Chen", + "Tao Zhang", + "Manxi Lin", + "Dunxian Huang", + "Teng Shi", + "Honghao Fu", + "Mengyan Li", + "Xinming Zhang", + "Chenchi Zhang", + "Xuan Lu", + "Xiaoxiong Du", + "Haibin Chen", + "Shaolin Ye", + "Hao Chang", + "Xiaoqi Li", + "Shuwen Xiao", + "Yujin Yuan", + "Jingxuan Feng", + "Shaopan Xiong", + "Huimin Yi", + "Ju Huang", + "Qiu Shen", + "Ying Chen", + "Junjun Zheng", + "Xiangheng Kong", + "Yuning Jiang" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.31693", + "source": "arxiv", + "source_id": "arxiv:2606.31693", + "pdf_url": "https://arxiv.org/pdf/2606.31693", + "primary_query": "llm-agent" + }, + { + "id": "2606.31252", + "title": "Embodied CAD: Solver-Grounded LLM Agents for Parametric B-Rep Assembly Modeling", + "url": "https://arxiv.org/abs/2606.31252", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Fumin Liu", + "Haoyu Zhou", + "Fei Hao", + "Lin Yang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.31252", + "source": "arxiv", + "source_id": "arxiv:2606.31252", + "pdf_url": "https://arxiv.org/pdf/2606.31252", + "primary_query": "llm-agent" + }, + { + "id": "2606.31174", + "title": "ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents", + "url": "https://arxiv.org/abs/2606.31174", + "published": "2026-06-30", + "updated": "2026-07-02", + "authors": [ + "Kaiwen Xiong", + "Haonian Ji", + "Shi Qiu", + "Zeyu Zheng", + "Cihang Xie", + "Xinyu Ye", + "Huaxiu Yao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.31174", + "source": "arxiv", + "source_id": "arxiv:2606.31174", + "pdf_url": "https://arxiv.org/pdf/2606.31174", + "primary_query": "llm-agent" + }, + { + "id": "2606.31046", + "title": "OpenLife: Toward Open-World Artificial Life with Autonomous LLM Agents", + "url": "https://arxiv.org/abs/2606.31046", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Atsushi Masumori", + "Itsuki Doi", + "Norihiro Maruyama", + "Ryosuke Takata", + "Takashi Ikegami" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.31046", + "source": "arxiv", + "source_id": "arxiv:2606.31046", + "pdf_url": "https://arxiv.org/pdf/2606.31046", + "primary_query": "llm-agent" + }, + { + "id": "2606.31650", + "title": "ECHO: Prune to act, trace to learn with selective turn memory in agentic RL", + "url": "https://arxiv.org/abs/2606.31650", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Zijun Xie", + "Binbin Zheng", + "Enlei Gong", + "Jihua Liu", + "Yuyang You", + "Lingfeng Liu", + "Jiayao Tang", + "Guanqun Zhao", + "Aoqi Hu", + "Zeyu Chen" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.31650", + "source": "arxiv", + "source_id": "arxiv:2606.31650", + "pdf_url": "https://arxiv.org/pdf/2606.31650", + "primary_query": "language-agent" + }, + { + "id": "2606.31639", + "title": "A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems", + "url": "https://arxiv.org/abs/2606.31639", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Seyed Bagher Hashemi Natanzi", + "Bo Tang" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.GT", + "cs.LO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.31639", + "source": "arxiv", + "source_id": "arxiv:2606.31639", + "pdf_url": "https://arxiv.org/pdf/2606.31639", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.30566", + "title": "Forensic Trajectory Signatures for Agent Memory Poisoning Detection", + "url": "https://arxiv.org/abs/2606.30566", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Jun Wen Leong" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "llm-agent" + ], + "arxiv_id": "2606.30566", + "source": "arxiv", + "source_id": "arxiv:2606.30566", + "pdf_url": "https://arxiv.org/pdf/2606.30566", + "primary_query": "agent-memory" + }, + { + "id": "2606.29914", + "title": "MemDelta: Controlled Baselines and Hidden Confounds in Agent Memory Evaluation", + "url": "https://arxiv.org/abs/2606.29914", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Kuan Wang" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2606.29914", + "source": "arxiv", + "source_id": "arxiv:2606.29914", + "pdf_url": "https://arxiv.org/pdf/2606.29914", + "primary_query": "agent-memory" + }, + { + "id": "2606.30555", + "title": "Linguistic Firewall: Geometry as Defense in Multi-Agent Systems Routing", + "url": "https://arxiv.org/abs/2606.30555", + "published": "2026-06-29", + "updated": "2026-07-05", + "authors": [ + "Dvir Alsheich", + "Adar Peleg", + "Ben Hagag", + "Rom Himelstein", + "Amit Levi", + "Avi Mendelson" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30555", + "source": "arxiv", + "source_id": "arxiv:2606.30555", + "pdf_url": "https://arxiv.org/pdf/2606.30555", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.30546", + "title": "MAS-Lab: A Specification-Driven Validation Framework for Reliable Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.30546", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Jordan Augé", + "Giovanna Carofiglio", + "Giulio Grassi", + "Jacques Samain" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30546", + "source": "arxiv", + "source_id": "arxiv:2606.30546", + "pdf_url": "https://arxiv.org/pdf/2606.30546", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.30259", + "title": "Multi-Agentic System Leveraging Open-Source LLMs to Mitigate Disinformation Threats", + "url": "https://arxiv.org/abs/2606.30259", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Sebastian Kula", + "Martin Tamajka" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30259", + "source": "arxiv", + "source_id": "arxiv:2606.30259", + "pdf_url": "https://arxiv.org/pdf/2606.30259", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29742", + "title": "MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition", + "url": "https://arxiv.org/abs/2606.29742", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Zishan Su", + "Junjie Huang", + "Shiwen Shan", + "Xingyan Chen", + "Hui Zeng", + "Yuxin Su", + "Yanlin Wang", + "Michael R. Lyu" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.29742", + "source": "arxiv", + "source_id": "arxiv:2606.29742", + "pdf_url": "https://arxiv.org/pdf/2606.29742", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29270", + "title": "Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates", + "url": "https://arxiv.org/abs/2606.29270", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Chuan He", + "Zebin Chen", + "Zhengyi Yang", + "Shaobo Qiao", + "Mingchen Ju", + "Jiate Liu", + "Dong Wen", + "Guanfeng Liu" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.29270", + "source": "arxiv", + "source_id": "arxiv:2606.29270", + "pdf_url": "https://arxiv.org/pdf/2606.29270", + "primary_query": "llm-agent" + }, + { + "id": "2606.29654", + "title": "Budgeted Act-or-Defer Multi-Agent LLM Deliberation with Local Reliability Bounds", + "url": "https://arxiv.org/abs/2606.29654", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Mengdie Flora Wang", + "Haochen Xie", + "Guanghui Wang", + "Devin Zhang", + "Jae Oh Woo" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.29654", + "source": "arxiv", + "source_id": "arxiv:2606.29654", + "pdf_url": "https://arxiv.org/pdf/2606.29654", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28958", + "title": "When Latent Agents Lie: KV-Cache Integrity in Multi-Agent LLM Collaboration", + "url": "https://arxiv.org/abs/2606.28958", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Luís Brito", + "Carlos Baquero" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-safety", + "memory", + "multi-agent", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.28958", + "source": "arxiv", + "source_id": "arxiv:2606.28958", + "pdf_url": "https://arxiv.org/pdf/2606.28958", + "primary_query": "llm-agent" + }, + { + "id": "2606.28666", + "title": "Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare", + "url": "https://arxiv.org/abs/2606.28666", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Liam Kearns" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "rag-agent" + ], + "arxiv_id": "2606.28666", + "source": "arxiv", + "source_id": "arxiv:2606.28666", + "pdf_url": "https://arxiv.org/pdf/2606.28666", + "primary_query": "agentic-ai" + }, + { + "id": "2606.27990", + "title": "AdvancedShelLM: A Stateful Multi-Agent LLM Honeypot for SSH Deception", + "url": "https://arxiv.org/abs/2606.27990", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Muris Sladić", + "Eman Alibalić", + "Veronica Valeros", + "Carlos Catania", + "Sebastian Garcia" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.27990", + "source": "arxiv", + "source_id": "arxiv:2606.27990", + "pdf_url": "https://arxiv.org/pdf/2606.27990", + "primary_query": "llm-agent" + }, + { + "id": "2606.27929", + "title": "When Multi-Robot Systems Meet Agentic AI:Towards Embodied Collective Intelligence", + "url": "https://arxiv.org/abs/2606.27929", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Yuxuan Yan", + "Yuanyuan Jia", + "Qianqian Yang" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "multi-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.27929", + "source": "arxiv", + "source_id": "arxiv:2606.27929", + "pdf_url": "https://arxiv.org/pdf/2606.27929", + "primary_query": "ai-agent" + }, + { + "id": "2606.28434", + "title": "SWE-MeM: Learning Adaptive Memory Management for Long-Horizon Coding Agents", + "url": "https://arxiv.org/abs/2606.28434", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Shuzheng Gao", + "Wenhao Zeng", + "Zhaojian Yu", + "Jianqiao Wangni", + "Chaozheng Wang", + "Kai Cai", + "Shilin He", + "Michael R. Lyu" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent", + "memory", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.28434", + "source": "arxiv", + "source_id": "arxiv:2606.28434", + "pdf_url": "https://arxiv.org/pdf/2606.28434", + "primary_query": "coding-agent" + }, + { + "id": "2606.28570", + "title": "Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG", + "url": "https://arxiv.org/abs/2606.28570", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Deep Ghosal", + "Ishani Sen", + "Wazib Ansar", + "Amlan Chakrabarti" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "rag-agent" + ], + "arxiv_id": "2606.28570", + "source": "arxiv", + "source_id": "arxiv:2606.28570", + "pdf_url": "https://arxiv.org/pdf/2606.28570", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28182", + "title": "LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior", + "url": "https://arxiv.org/abs/2606.28182", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Qinhong Zhou", + "Chuang Gan", + "Anoop Cherian" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CV", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28182", + "source": "arxiv", + "source_id": "arxiv:2606.28182", + "pdf_url": "https://arxiv.org/pdf/2606.28182", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.25514", + "title": "Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution", + "url": "https://arxiv.org/abs/2606.25514", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yang Chen", + "Aliya Ahmad", + "Yiheng Zhou", + "Reyhaneh Jabbarvand" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.25514", + "source": "arxiv", + "source_id": "arxiv:2606.25514", + "pdf_url": "https://arxiv.org/pdf/2606.25514", + "primary_query": "coding-agent" + }, + { + "id": "2606.25588", + "title": "IntentTester: Intent-Driven Multi-agent Framework for Cross-Library Test Migration", + "url": "https://arxiv.org/abs/2606.25588", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yi Gao", + "Ziyuan Zhang", + "Xing Hu", + "Xiaohu Yang", + "Xin Xia" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25588", + "source": "arxiv", + "source_id": "arxiv:2606.25588", + "pdf_url": "https://arxiv.org/pdf/2606.25588", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.25400", + "title": "BrainAgent: A Large Language Model-Driven Multi-Agent Framework for Autonomous Brain Signal Understanding", + "url": "https://arxiv.org/abs/2606.25400", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yangxuan Zhou", + "Sha Zhao", + "Jiquan Wang", + "Shijian Li", + "Gang Pan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25400", + "source": "arxiv", + "source_id": "arxiv:2606.25400", + "pdf_url": "https://arxiv.org/pdf/2606.25400", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24775", + "title": "Are We Ready For An Agent-Native Memory System?", + "url": "https://arxiv.org/abs/2606.24775", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Wei Zhou", + "Xuanhe Zhou", + "Shaokun Han", + "Hongming Xu", + "Guoliang Li", + "Zhiyu Li", + "Feiyu Xiong", + "Fan Wu" + ], + "categories": [ + "cs.CL", + "cs.DB", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.24775", + "source": "arxiv", + "source_id": "arxiv:2606.24775", + "pdf_url": "https://arxiv.org/pdf/2606.24775", + "primary_query": "agent-memory" + }, + { + "id": "2606.24649", + "title": "Agentic Collaborative Cognition for Zero-Shot 3D Understanding", + "url": "https://arxiv.org/abs/2606.24649", + "published": "2026-06-23", + "updated": "2026-06-25", + "authors": [ + "Wenxin Wang", + "Bo Zhang", + "Feng Chen", + "Zixuan Wang", + "Wen Li", + "Changsheng Li", + "Yinjie Lei" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24649", + "source": "arxiv", + "source_id": "arxiv:2606.24649", + "pdf_url": "https://arxiv.org/pdf/2606.24649", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24437", + "title": "ReM-MoA: Reasoning Memory Sustains Mixture-of-Agents Scaling", + "url": "https://arxiv.org/abs/2606.24437", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Heng Ping", + "Arijit Bhattacharjee", + "Peiyu Zhang", + "Shixuan Li", + "Wei Yang", + "Ali Jannesari", + "Nesreen Ahmed", + "Paul Bogdan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24437", + "source": "arxiv", + "source_id": "arxiv:2606.24437", + "pdf_url": "https://arxiv.org/pdf/2606.24437", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22647", + "title": "RAVEN: Agentic RAG for Automated Vulnerability Repair", + "url": "https://arxiv.org/abs/2606.22647", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Varun Gadey", + "Zijie Liu", + "Alexandra Dmitrienko" + ], + "categories": [ + "cs.CR", + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.22647", + "source": "arxiv", + "source_id": "arxiv:2606.22647", + "pdf_url": "https://arxiv.org/pdf/2606.22647", + "primary_query": "rag-agent" + }, + { + "id": "2606.22330", + "title": "Hypothesis-Driven Skill Optimization for LLM Agents", + "url": "https://arxiv.org/abs/2606.22330", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Fangxin Shang", + "Yehui Yang" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-safety", + "coding-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.22330", + "source": "arxiv", + "source_id": "arxiv:2606.22330", + "pdf_url": "https://arxiv.org/pdf/2606.22330", + "primary_query": "planning-agent" + }, + { + "id": "2606.21740", + "title": "Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents", + "url": "https://arxiv.org/abs/2606.21740", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Rajesh Mangannavar", + "Zachary Coalson", + "Pranay Dugar", + "Prasad Tadepalli" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.21740", + "source": "arxiv", + "source_id": "arxiv:2606.21740", + "pdf_url": "https://arxiv.org/pdf/2606.21740", + "primary_query": "planning-agent" + }, + { + "id": "2606.18467", + "title": "ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift", + "url": "https://arxiv.org/abs/2606.18467", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Jeffery Opoku", + "David Banahene" + ], + "categories": [ + "stat.ML", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "agentic-ai", + "ai-agent", + "rag-agent", + "tool-use" + ], + "arxiv_id": "2606.18467", + "source": "arxiv", + "source_id": "arxiv:2606.18467", + "pdf_url": "https://arxiv.org/pdf/2606.18467", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18142", + "title": "Your AI Travel Agent Would Book You a Bullfight: An Agentic Benchmark for Implicit Animal Welfare in Frontier AI Models", + "url": "https://arxiv.org/abs/2606.18142", + "published": "2026-06-16", + "updated": "2026-07-06", + "authors": [ + "Jasmine Brazilek", + "Joel Christoph", + "Maheep Chaudhary", + "Oliver Tullio", + "Carol Kline", + "Miles Tidmarsh", + "Arturs Kanepajs" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.18142", + "source": "arxiv", + "source_id": "arxiv:2606.18142", + "pdf_url": "https://arxiv.org/pdf/2606.18142", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.16591", + "title": "SING: Synthetic Intention Graph for Scalable Active Tool Discovery in LLM Agents", + "url": "https://arxiv.org/abs/2606.16591", + "published": "2026-06-15", + "updated": "2026-06-16", + "authors": [ + "Qiao Xiao", + "Haochen Shi", + "Yisen Gao", + "Wenbin Hu", + "Huihao Jing", + "Tianshi Zheng", + "Baixuan Xu", + "Ziheng Zhang", + "Weiqi Wang", + "Haoran Li", + "Jiaxin Bai", + "Yangqiu Song" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.16591", + "source": "arxiv", + "source_id": "arxiv:2606.16591", + "pdf_url": "https://arxiv.org/pdf/2606.16591", + "primary_query": "tool-use" + }, + { + "id": "2606.15684", + "title": "Multi-agent Framework for Time-Sensitive Complementary Collaboration in Minecraft", + "url": "https://arxiv.org/abs/2606.15684", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Juheon Yi", + "Jinglu Wang", + "Xiaoyi Zhang", + "Yan Lu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.15684", + "source": "arxiv", + "source_id": "arxiv:2606.15684", + "pdf_url": "https://arxiv.org/pdf/2606.15684", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.15931", + "title": "DeepRoot: A KG-Coordinated Multi-Agent System for Therapeutic Reasoning over Historical Medical Texts", + "url": "https://arxiv.org/abs/2606.15931", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Zijian Carl Ma", + "Sean J. Wang", + "Sijbren Kramer", + "Li Erran Li" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.15931", + "source": "arxiv", + "source_id": "arxiv:2606.15931", + "pdf_url": "https://arxiv.org/pdf/2606.15931", + "primary_query": "tool-use" + }, + { + "id": "2606.12703", + "title": "SMSR: Certified Defence Against Runtime Memory Poisoning in Persistent LLM Agent Systems", + "url": "https://arxiv.org/abs/2606.12703", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Tarun Sharma" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.12703", + "source": "arxiv", + "source_id": "arxiv:2606.12703", + "pdf_url": "https://arxiv.org/pdf/2606.12703", + "primary_query": "rag-agent" + }, + { + "id": "2606.10684", + "title": "Divide and Cooperate: Role-Decomposed Multi-Agent LLM Training with Cross-Agent Learning Signals", + "url": "https://arxiv.org/abs/2606.10684", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Jaewan Park", + "Solbee Cho", + "Jay-Yoon Lee" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.10684", + "source": "arxiv", + "source_id": "arxiv:2606.10684", + "pdf_url": "https://arxiv.org/pdf/2606.10684", + "primary_query": "language-agent" + }, + { + "id": "2606.10616", + "title": "Learning What to Remember: Observability-Safe Memory Retention via Constrained Optimization for Long-Horizon Language Agents", + "url": "https://arxiv.org/abs/2606.10616", + "published": "2026-06-09", + "updated": "2026-06-29", + "authors": [ + "Qingcan Kang", + "Liu Mingyang", + "Shixiong Kai", + "Kaichao Liang", + "Tao Zhong", + "Mingxuan Yuan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.10616", + "source": "arxiv", + "source_id": "arxiv:2606.10616", + "pdf_url": "https://arxiv.org/pdf/2606.10616", + "primary_query": "language-agent" + }, + { + "id": "2606.10933", + "title": "Frontier Coding Agents Use Metaprogramming to Adapt to Unfamiliar Programming Languages", + "url": "https://arxiv.org/abs/2606.10933", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Aman Sharma", + "Sushrut Thorat", + "Paras Chopra" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.10933", + "source": "arxiv", + "source_id": "arxiv:2606.10933", + "pdf_url": "https://arxiv.org/pdf/2606.10933", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.10577", + "title": "AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness", + "url": "https://arxiv.org/abs/2606.10577", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yijian Li", + "Changze Li", + "Hantian Shi", + "Jiaying Luo", + "Jiyuan Cai", + "Ming Yang", + "Tong Qin" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.10577", + "source": "arxiv", + "source_id": "arxiv:2606.10577", + "pdf_url": "https://arxiv.org/pdf/2606.10577", + "primary_query": "agent-memory" + }, + { + "id": "2606.10304", + "title": "MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents", + "url": "https://arxiv.org/abs/2606.10304", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Pratibha Revankar", + "Kargi Chauhan", + "Jihye Kim", + "Sadiba Nusrat Nur", + "Vincent Siu", + "Chenguang Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.10304", + "source": "arxiv", + "source_id": "arxiv:2606.10304", + "pdf_url": "https://arxiv.org/pdf/2606.10304", + "primary_query": "planning-agent" + }, + { + "id": "2606.07682", + "title": "SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work?", + "url": "https://arxiv.org/abs/2606.07682", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Rishi Desai", + "Jesse Hu", + "Joan Cabezas", + "Neel Harsola", + "Pratyush Shukla", + "Roey Ben Chaim", + "Adnan El Assadi", + "Omkaar Mukund Kamath", + "Fenil Faldu", + "Prannay Hebbar", + "Jiankai Sun", + "Yiyuan Li", + "Pramod Srinivasan", + "Ishan Gupta", + "Christopher Settles", + "Daniel Wang", + "Derek Chen", + "Pranav Raja", + "Albert Liu", + "Marek Šuppa", + "Nevasini Sasikumar", + "Luyang Kong", + "Erik Quintanilla", + "Xiangyi Li", + "Ivan Bercovich", + "Steven Dillmann" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.07682", + "source": "arxiv", + "source_id": "arxiv:2606.07682", + "pdf_url": "https://arxiv.org/pdf/2606.07682", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.07867", + "title": "The Cold-Start Safety Gap in LLM Agents", + "url": "https://arxiv.org/abs/2606.07867", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Chung-En Sun", + "Linbo Liu", + "Tsui-Wei Weng" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.07867", + "source": "arxiv", + "source_id": "arxiv:2606.07867", + "pdf_url": "https://arxiv.org/pdf/2606.07867", + "primary_query": "agent-safety" + }, + { + "id": "2606.07711", + "title": "Rosetta Memory: Adaptive Memory for Cross-LLM Agents", + "url": "https://arxiv.org/abs/2606.07711", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Hao Yang", + "Shiqi Shen", + "Haoxuan Li", + "Zhipeng Wang", + "Zhi Gong", + "Xu Chen" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "coding-agent", + "memory", + "planning", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.07711", + "source": "arxiv", + "source_id": "arxiv:2606.07711", + "pdf_url": "https://arxiv.org/pdf/2606.07711", + "primary_query": "planning-agent" + }, + { + "id": "2606.06054", + "title": "Beyond Similarity: Trustworthy Memory Search for Personal AI Agents", + "url": "https://arxiv.org/abs/2606.06054", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Jiawen Zhang", + "Kejia Chen", + "Jiachen Ma", + "Yangfan Hu", + "Lipeng He", + "Yechao Zhang", + "Jian Liu", + "Xiaohu Yang", + "Tianwei Zhang", + "Ruoxi Jia" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.06054", + "source": "arxiv", + "source_id": "arxiv:2606.06054", + "pdf_url": "https://arxiv.org/pdf/2606.06054", + "primary_query": "agent-memory" + }, + { + "id": "2606.05805", + "title": "From Risk Classification to Action Plan Remediation: A Guardrail Feedback Driven Framework for LLM Agents", + "url": "https://arxiv.org/abs/2606.05805", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Yuhao Sun", + "Jiacheng Zhang", + "Shaanan Cohney", + "Zhexin Zhang", + "Feng Liu", + "Xingliang Yuan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.05805", + "source": "arxiv", + "source_id": "arxiv:2606.05805", + "pdf_url": "https://arxiv.org/pdf/2606.05805", + "primary_query": "planning-agent" + }, + { + "id": "2606.04599", + "title": "Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection", + "url": "https://arxiv.org/abs/2606.04599", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Yongzi Yu", + "Ao Li", + "Le Wang", + "Ziyue Li", + "Fugee Tsung", + "Yuxuan Liang", + "Man Li" + ], + "categories": [ + "cs.AI", + "cs.CE" + ], + "topics": [ + "agent-safety", + "multi-agent", + "planning", + "rag", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.04599", + "source": "arxiv", + "source_id": "arxiv:2606.04599", + "pdf_url": "https://arxiv.org/pdf/2606.04599", + "primary_query": "planning-agent" + }, + { + "id": "2606.04051", + "title": "RUBAS: Rubric-Based Reinforcement Learning for Agent Safety", + "url": "https://arxiv.org/abs/2606.04051", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Xian Qi Loye", + "Qinglin Su", + "Zhexin Zhang", + "Shiyao Cui", + "Qi Zhu", + "Fei Mi", + "Hongning Wang", + "Minlie Huang" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.04051", + "source": "arxiv", + "source_id": "arxiv:2606.04051", + "pdf_url": "https://arxiv.org/pdf/2606.04051", + "primary_query": "agent-safety" + }, + { + "id": "2606.02812", + "title": "Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection", + "url": "https://arxiv.org/abs/2606.02812", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Sihang Zeng", + "Matthew Thompson", + "Ruth Etzioni", + "Meliha Yetisgen" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "rag", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.02812", + "source": "arxiv", + "source_id": "arxiv:2606.02812", + "pdf_url": "https://arxiv.org/pdf/2606.02812", + "primary_query": "agent-memory" + }, + { + "id": "2606.01528", + "title": "Joint Agent Memory and Exploration Learning via Novelty Signals", + "url": "https://arxiv.org/abs/2606.01528", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Shizuo Tian", + "Xiaohong Weng", + "Rui Kong", + "Yuxuan Chen", + "Guohong Liu", + "Yuebing Song", + "Jiacheng Liu", + "Yuchen Li", + "Dawei Yin", + "Ting Cao", + "Yunxin Liu", + "Yuanchun Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.01528", + "source": "arxiv", + "source_id": "arxiv:2606.01528", + "pdf_url": "https://arxiv.org/pdf/2606.01528", + "primary_query": "agent-memory" + }, + { + "id": "2606.01041", + "title": "ExpWeaver: LLM Agents Learn from Experience via Latent RAG", + "url": "https://arxiv.org/abs/2606.01041", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Tao Feng", + "Tianyang Luo", + "Jingjun Xu", + "Zhigang Hua", + "Yan Xie", + "Shuang Yang", + "Ge Liu", + "Jiaxuan You" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent", + "rag-agent" + ], + "arxiv_id": "2606.01041", + "source": "arxiv", + "source_id": "arxiv:2606.01041", + "pdf_url": "https://arxiv.org/pdf/2606.01041", + "primary_query": "planning-agent" + }, + { + "id": "2605.30907", + "title": "BlueFin: Benchmarking LLM Agents on Financial Spreadsheets", + "url": "https://arxiv.org/abs/2605.30907", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Srivatsa Kundurthy", + "Clara Na", + "Colton Moraine", + "Anoushka Mohta", + "Case Winter", + "George Fang", + "John Ling", + "Emma Strubell", + "Zach Kirshner" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.30907", + "source": "arxiv", + "source_id": "arxiv:2605.30907", + "pdf_url": "https://arxiv.org/pdf/2605.30907", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.30858", + "title": "ForecastCompass: Guiding Agentic Forecasting with Adaptive Factor Memory", + "url": "https://arxiv.org/abs/2605.30858", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Yurui Chang", + "Yongkang Du", + "Yuanpu Cao", + "Jinghui Chen", + "Lu Lin" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.30858", + "source": "arxiv", + "source_id": "arxiv:2605.30858", + "pdf_url": "https://arxiv.org/pdf/2605.30858", + "primary_query": "agent-memory" + }, + { + "id": "2605.30058", + "title": "HEART-Bench: Do LLM Agents Exhibit Human-like Psychology?", + "url": "https://arxiv.org/abs/2605.30058", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Weihan Peng", + "Chenxu Zhang", + "Qianao Wang", + "Yuling Shi", + "Heng Lian", + "Qihong Mao", + "Jiahao Pang", + "Chunliang Feng", + "Bowen Li", + "Xiaodong Gu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.30058", + "source": "arxiv", + "source_id": "arxiv:2605.30058", + "pdf_url": "https://arxiv.org/pdf/2605.30058", + "primary_query": "planning-agent" + }, + { + "id": "2605.27762", + "title": "PEAM: Parametric Embodied Agent Memory through Contrastive Internalization of Experience in Minecraft", + "url": "https://arxiv.org/abs/2605.27762", + "published": "2026-05-26", + "updated": "2026-06-01", + "authors": [ + "Yuchen Guo", + "Junli Gong", + "Weicheng Wang", + "Hongmin Cai", + "Yiu-ming Cheung", + "Weifeng Su" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.27762", + "source": "arxiv", + "source_id": "arxiv:2605.27762", + "pdf_url": "https://arxiv.org/pdf/2605.27762", + "primary_query": "agent-memory" + }, + { + "id": "2605.24659", + "title": "IterInject: Indirect Prompt Injection Against LLM Agents via Feedback-Guided Iterative Optimization", + "url": "https://arxiv.org/abs/2605.24659", + "published": "2026-05-23", + "updated": "2026-05-23", + "authors": [ + "Zixuan Chen", + "Jiaxiang Chen", + "Li Luo", + "Ke Xu", + "Xiaoxiang Huang", + "Tanfeng Sun", + "Xinghao Jiang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.24659", + "source": "arxiv", + "source_id": "arxiv:2605.24659", + "pdf_url": "https://arxiv.org/pdf/2605.24659", + "primary_query": "planning-agent" + }, + { + "id": "2605.23723", + "title": "MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection", + "url": "https://arxiv.org/abs/2605.23723", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Zhewen Tan", + "Yilun Yao", + "Huiyan Jin", + "Wenhan Yu", + "Guoan Wang", + "Mengyuan Fan", + "liang lu", + "Feng Liu", + "Xiangzheng Zhang", + "Duohe Ma", + "Tong Yang", + "Lin Sun" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.23723", + "source": "arxiv", + "source_id": "arxiv:2605.23723", + "pdf_url": "https://arxiv.org/pdf/2605.23723", + "primary_query": "agent-memory" + }, + { + "id": "2605.24219", + "title": "Beyond Final Answers: Auditing Trajectory-Level Hallucinations in Multi-Agent Industrial Workflows", + "url": "https://arxiv.org/abs/2605.24219", + "published": "2026-05-22", + "updated": "2026-05-26", + "authors": [ + "Harshada Badave", + "Santosh Borse", + "Andrea Gomez", + "Harshitha Narahari", + "Sara Carter", + "Vishwa Bhatt", + "Aishani Rachakonda", + "Shuxin Lin", + "Dhaval Patel" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.24219", + "source": "arxiv", + "source_id": "arxiv:2605.24219", + "pdf_url": "https://arxiv.org/pdf/2605.24219", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.22154", + "title": "IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents", + "url": "https://arxiv.org/abs/2605.22154", + "published": "2026-05-21", + "updated": "2026-05-21", + "authors": [ + "Daewon Choi", + "Kyunghyun Park", + "Woomin Song", + "Saket Dingliwal", + "Sai Muralidhar Jayanthi", + "Jinwoo Shin", + "Aram Galstyan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.22154", + "source": "arxiv", + "source_id": "arxiv:2605.22154", + "pdf_url": "https://arxiv.org/pdf/2605.22154", + "primary_query": "planning-agent" + }, + { + "id": "2605.21240", + "title": "APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents", + "url": "https://arxiv.org/abs/2605.21240", + "published": "2026-05-20", + "updated": "2026-05-20", + "authors": [ + "Yibo Li", + "Jiashuo Yang", + "Zhi Zheng", + "Zhiyuan Hu", + "Yuan Sui", + "Shizun Wang", + "Yufei He", + "Bryan Hooi" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.21240", + "source": "arxiv", + "source_id": "arxiv:2605.21240", + "pdf_url": "https://arxiv.org/pdf/2605.21240", + "primary_query": "planning-agent" + }, + { + "id": "2605.18930", + "title": "OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences", + "url": "https://arxiv.org/abs/2605.18930", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Kaixiang Wang", + "Jiong Lou", + "Zhaojiacheng Zhou", + "Jie Li" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.18930", + "source": "arxiv", + "source_id": "arxiv:2605.18930", + "pdf_url": "https://arxiv.org/pdf/2605.18930", + "primary_query": "agent-memory" + }, + { + "id": "2605.18284", + "title": "CommitDistill: A Lightweight Knowledge-Centric Memory Layer for Software Repositories", + "url": "https://arxiv.org/abs/2605.18284", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Divya Chukkapalli", + "Thejesh Avula", + "Aditya Aggarwal", + "Harsimran Singh", + "Amith Tallanki" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.18284", + "source": "arxiv", + "source_id": "arxiv:2605.18284", + "pdf_url": "https://arxiv.org/pdf/2605.18284", + "primary_query": "agent-memory" + }, + { + "id": "2605.14421", + "title": "MemLineage: Lineage-Guided Enforcement for LLM Agent Memory", + "url": "https://arxiv.org/abs/2605.14421", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Ciyan Ouyang", + "Rui Hou" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.14421", + "source": "arxiv", + "source_id": "arxiv:2605.14421", + "pdf_url": "https://arxiv.org/pdf/2605.14421", + "primary_query": "agent-memory" + }, + { + "id": "2605.14892", + "title": "Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems", + "url": "https://arxiv.org/abs/2605.14892", + "published": "2026-05-14", + "updated": "2026-05-15", + "authors": [ + "Shihao Qi", + "Jie Ma", + "Rui Xing", + "Wei Guo", + "Xiao Huang", + "Zhitao Gao", + "Jianhao Deng", + "Jun Liu", + "Lingling Zhang", + "Bifan Wei", + "Boqian Yang", + "Pinghui Wang", + "Jianwen Sun", + "Jing Tao", + "Yaqiang Wu", + "Hui Liu", + "Yu Yao", + "Tongliang Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14892", + "source": "arxiv", + "source_id": "arxiv:2605.14892", + "pdf_url": "https://arxiv.org/pdf/2605.14892", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.14527", + "title": "Lang2MLIP: End-to-End Language-to-Machine Learning Interatomic Potential Development with Autonomous Agentic Workflows", + "url": "https://arxiv.org/abs/2605.14527", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Wenwen Li", + "Yuki Orimo", + "Nontawat Charoenphakdee" + ], + "categories": [ + "cs.LG", + "cond-mat.mtrl-sci", + "physics.comp-ph" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14527", + "source": "arxiv", + "source_id": "arxiv:2605.14527", + "pdf_url": "https://arxiv.org/pdf/2605.14527", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.13481", + "title": "PersonalAI 2.0: Enhancing knowledge graph traversal/retrieval with planning mechanism for Personalized LLM Agents", + "url": "https://arxiv.org/abs/2605.13481", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Mikhail Menschikov", + "Matvey Iskornev", + "Alexander Kharitonov", + "Alina Bogdanova", + "Mikhail Belkin", + "Ekaterina Lisitsyna", + "Artyom Sosedka", + "Victoria Dochkina", + "Ruslan Kostoev", + "Ilia Perepechkin", + "Evgeny Burnaev" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.13481", + "source": "arxiv", + "source_id": "arxiv:2605.13481", + "pdf_url": "https://arxiv.org/pdf/2605.13481", + "primary_query": "planning-agent" + }, + { + "id": "2605.12260", + "title": "PRISM: Pareto-Efficient Retrieval over Intent-Aware Structured Memory for Long-Horizon Agents", + "url": "https://arxiv.org/abs/2605.12260", + "published": "2026-05-12", + "updated": "2026-05-22", + "authors": [ + "Jingyi Peng", + "Zhongwei Wan", + "Weiting Liu", + "Qiuzhuang Sun" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.12260", + "source": "arxiv", + "source_id": "arxiv:2605.12260", + "pdf_url": "https://arxiv.org/pdf/2605.12260", + "primary_query": "language-agent" + }, + { + "id": "2605.11225", + "title": "PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement", + "url": "https://arxiv.org/abs/2605.11225", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Tuo Zhang", + "Alin-Ionut Popa", + "Yan Xu", + "Rui Song", + "Dimitrios Dimitriadis" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2605.11225", + "source": "arxiv", + "source_id": "arxiv:2605.11225", + "pdf_url": "https://arxiv.org/pdf/2605.11225", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.09692", + "title": "Causal state binding predicts action control in language agents", + "url": "https://arxiv.org/abs/2605.09692", + "published": "2026-05-10", + "updated": "2026-06-01", + "authors": [ + "Xiao Jia" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.09692", + "source": "arxiv", + "source_id": "arxiv:2605.09692", + "pdf_url": "https://arxiv.org/pdf/2605.09692", + "primary_query": "language-agent" + }, + { + "id": "2605.07251", + "title": "Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning", + "url": "https://arxiv.org/abs/2605.07251", + "published": "2026-05-08", + "updated": "2026-05-08", + "authors": [ + "Yuyang Wu", + "Yue Huang", + "Shuaike Shen", + "Xujian Wang", + "Shuhao Zhang", + "Qiyao Xue", + "Weichen Liu", + "Runtian Gao", + "Jian Ma", + "Xiangliang Zhang", + "Olexandr Isayev" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.07251", + "source": "arxiv", + "source_id": "arxiv:2605.07251", + "pdf_url": "https://arxiv.org/pdf/2605.07251", + "primary_query": "planning-agent" + }, + { + "id": "2605.06713", + "title": "Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand", + "url": "https://arxiv.org/abs/2605.06713", + "published": "2026-05-06", + "updated": "2026-05-06", + "authors": [ + "Christopher Koch" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-safety", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "planning-agent" + ], + "arxiv_id": "2605.06713", + "source": "arxiv", + "source_id": "arxiv:2605.06713", + "pdf_url": "https://arxiv.org/pdf/2605.06713", + "primary_query": "agent-safety" + }, + { + "id": "2605.02240", + "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments", + "url": "https://arxiv.org/abs/2605.02240", + "published": "2026-05-04", + "updated": "2026-05-04", + "authors": [ + "Ruoqi Liu", + "Imran Q. Mohiuddin", + "Austin J. Schoeffler", + "Kavita Renduchintala", + "Ashwin Nayak", + "Prasantha L. Vemu", + "Shivam C. Vedak", + "Kameron C. Black", + "John L. Havlik", + "Isaac Ogunmola", + "Stephen P. Ma", + "Roopa Dhatt", + "Jonathan H. Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.02240", + "source": "arxiv", + "source_id": "arxiv:2605.02240", + "pdf_url": "https://arxiv.org/pdf/2605.02240", + "primary_query": "planning-agent" + }, + { + "id": "2604.11557", + "title": "UniToolCall: Unifying Tool-Use Representation, Data, and Evaluation for LLM Agents", + "url": "https://arxiv.org/abs/2604.11557", + "published": "2026-04-13", + "updated": "2026-05-25", + "authors": [ + "Yijuan Liang", + "Xinghao Chen", + "Yifan Ge", + "Ziyi Wu", + "Hao Wu", + "Changyu Zeng", + "Wei Xing", + "Xiaoyu Shen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2604.11557", + "source": "arxiv", + "source_id": "arxiv:2604.11557", + "pdf_url": "https://arxiv.org/pdf/2604.11557", + "primary_query": "function-calling" + }, + { + "id": "2604.10577", + "title": "The Blind Spot of Agent Safety: How Benign User Instructions Expose Critical Vulnerabilities in Computer-Use Agents", + "url": "https://arxiv.org/abs/2604.10577", + "published": "2026-04-12", + "updated": "2026-04-17", + "authors": [ + "Xuwei Ding", + "Skylar Zhai", + "Linxin Song", + "Jiate Li", + "Taiwei Shi", + "Nicholas Meade", + "Siva Reddy", + "Jian Kang", + "Jieyu Zhao" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.10577", + "source": "arxiv", + "source_id": "arxiv:2604.10577", + "pdf_url": "https://arxiv.org/pdf/2604.10577", + "primary_query": "agent-safety" + }, + { + "id": "2603.24257", + "title": "Memory-Augmented Vision-Language Agents for Persistent and Semantically Consistent Object Captioning", + "url": "https://arxiv.org/abs/2603.24257", + "published": "2026-03-25", + "updated": "2026-03-30", + "authors": [ + "Tommaso Galliena", + "Stefano Rosa", + "Tommaso Apicella", + "Pietro Morerio", + "Alessio Del Bue", + "Lorenzo Natale" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.24257", + "source": "arxiv", + "source_id": "arxiv:2603.24257", + "pdf_url": "https://arxiv.org/pdf/2603.24257", + "primary_query": "language-agent" + }, + { + "id": "2603.10492", + "title": "Human-AI Co-reasoning for Clinical Diagnosis with Evidence-Integrated Language Agent", + "url": "https://arxiv.org/abs/2603.10492", + "published": "2026-03-11", + "updated": "2026-03-18", + "authors": [ + "Zhongzhen Huang", + "Yan Ling", + "Hong Chen", + "Ye Feng", + "Li Wu", + "Linjie Mu", + "Shaoting Zhang", + "Xiaofan Zhang", + "Kun Qian", + "Xiaomu Li" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.10492", + "source": "arxiv", + "source_id": "arxiv:2603.10492", + "pdf_url": "https://arxiv.org/pdf/2603.10492", + "primary_query": "language-agent" + }, + { + "id": "2603.07496", + "title": "From Thinker to Society: Security in Hierarchical Autonomy Evolution of AI Agents", + "url": "https://arxiv.org/abs/2603.07496", + "published": "2026-03-08", + "updated": "2026-03-21", + "authors": [ + "Xiaolei Zhang", + "Lu Zhou", + "Xiaogang Xu", + "Jiafei Wu", + "Tianyu Du", + "Heqing Huang", + "Hao Peng", + "Zhe Liu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.07496", + "source": "arxiv", + "source_id": "arxiv:2603.07496", + "pdf_url": "https://arxiv.org/pdf/2603.07496", + "primary_query": "agent-safety" + }, + { + "id": "2603.03680", + "title": "MAGE: Meta-Reinforcement Learning for Language Agents toward Strategic Exploration and Exploitation", + "url": "https://arxiv.org/abs/2603.03680", + "published": "2026-03-04", + "updated": "2026-03-04", + "authors": [ + "Lu Yang", + "Zelai Xu", + "Minyang Xie", + "Jiaxuan Gao", + "Zhao Shok", + "Yu Wang", + "Yi Wu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.03680", + "source": "arxiv", + "source_id": "arxiv:2603.03680", + "pdf_url": "https://arxiv.org/pdf/2603.03680", + "primary_query": "language-agent" + }, + { + "id": "2603.02711", + "title": "A Natural Language Agentic Approach to Study Affective Polarization", + "url": "https://arxiv.org/abs/2603.02711", + "published": "2026-03-03", + "updated": "2026-03-03", + "authors": [ + "Stephanie Anneris Malvicini", + "Ewelina Gajewska", + "Arda Derbent", + "Katarzyna Budzynska", + "Jarosław A. Chudziak", + "Maria Vanina Martinez" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "multi-agent", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.02711", + "source": "arxiv", + "source_id": "arxiv:2603.02711", + "pdf_url": "https://arxiv.org/pdf/2603.02711", + "primary_query": "language-agent" + }, + { + "id": "2603.03515", + "title": "The Controllability Trap: A Governance Framework for Military AI Agents", + "url": "https://arxiv.org/abs/2603.03515", + "published": "2026-03-03", + "updated": "2026-03-03", + "authors": [ + "Subramanyam Sahoo" + ], + "categories": [ + "cs.CY", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "world-model" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.03515", + "source": "arxiv", + "source_id": "arxiv:2603.03515", + "pdf_url": "https://arxiv.org/pdf/2603.03515", + "primary_query": "agent-safety" + }, + { + "id": "2604.03242", + "title": "DRAFT: Task Decoupled Latent Reasoning for Agent Safety", + "url": "https://arxiv.org/abs/2604.03242", + "published": "2026-02-11", + "updated": "2026-02-11", + "authors": [ + "Lin Wang", + "Junfeng Fang", + "Dan Zhang", + "Fei Shen", + "Xiang Wang", + "Tat-Seng Chua" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.03242", + "source": "arxiv", + "source_id": "arxiv:2604.03242", + "pdf_url": "https://arxiv.org/pdf/2604.03242", + "primary_query": "agent-safety" + }, + { + "id": "2603.08721", + "title": "KernelCraft: Benchmarking for Agentic Close-to-Metal Kernel Generation on Emerging Hardware", + "url": "https://arxiv.org/abs/2603.08721", + "published": "2026-02-10", + "updated": "2026-05-29", + "authors": [ + "Jiayi Nie", + "Haoran Wu", + "Yao Lai", + "Zeyu Cao", + "Cheng Zhang", + "Binglei Lou", + "Erwei Wang", + "Jianyi Cheng", + "Timothy M. Jones", + "Robert Mullins", + "Rika Antonova", + "Yiren Zhao" + ], + "categories": [ + "cs.AR", + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.08721", + "source": "arxiv", + "source_id": "arxiv:2603.08721", + "pdf_url": "https://arxiv.org/pdf/2603.08721", + "primary_query": "function-calling" + }, + { + "id": "2602.03224", + "title": "TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking", + "url": "https://arxiv.org/abs/2602.03224", + "published": "2026-02-03", + "updated": "2026-06-06", + "authors": [ + "Yu Cheng", + "Yongkang Hu", + "Jiuan Zhou", + "Yushuo Zhang", + "Yihang Chen", + "Huichi Zhou", + "Mingang Chen", + "Zhizhong Zhang", + "Kun Shao", + "Yuan Xie", + "Zhaoxia Yin" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.03224", + "source": "arxiv", + "source_id": "arxiv:2602.03224", + "pdf_url": "https://arxiv.org/pdf/2602.03224", + "primary_query": "agent-safety" + }, + { + "id": "2512.11682", + "title": "MedAI: Evaluating TxAgent's Therapeutic Agentic Reasoning in the NeurIPS CURE-Bench Competition", + "url": "https://arxiv.org/abs/2512.11682", + "published": "2025-12-12", + "updated": "2026-06-15", + "authors": [ + "Tim Cofala", + "Christian Kalfar", + "Jingge Xiao", + "Johanna Schrader", + "Michelle Tang", + "Wolfgang Nejdl" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.11682", + "source": "arxiv", + "source_id": "arxiv:2512.11682", + "pdf_url": "https://arxiv.org/pdf/2512.11682", + "primary_query": "function-calling" + }, + { + "id": "2511.15203", + "title": "Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks", + "url": "https://arxiv.org/abs/2511.15203", + "published": "2025-11-19", + "updated": "2025-11-19", + "authors": [ + "Zimo Ji", + "Xunguang Wang", + "Zongjie Li", + "Pingchuan Ma", + "Yudong Gao", + "Daoyuan Wu", + "Xincheng Yan", + "Tian Tian", + "Shuai Wang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2511.15203", + "source": "arxiv", + "source_id": "arxiv:2511.15203", + "pdf_url": "https://arxiv.org/pdf/2511.15203", + "primary_query": "function-calling" + }, + { + "id": "2510.18586", + "title": "TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications", + "url": "https://arxiv.org/abs/2510.18586", + "published": "2025-10-21", + "updated": "2026-05-20", + "authors": [ + "Zhuohang Bian", + "Feiyang Wu", + "Zhuoran Li", + "Teng Ma", + "Youwei Zhuo" + ], + "categories": [ + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.18586", + "source": "arxiv", + "source_id": "arxiv:2510.18586", + "pdf_url": "https://arxiv.org/pdf/2510.18586", + "primary_query": "function-calling" + }, + { + "id": "2607.06001", + "title": "Information Limits and Attractor Dynamics in Economies of Frontier LLM Agents: A Pre-Registered Test", + "url": "https://arxiv.org/abs/2607.06001", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Cheng Qian" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.06001", + "source": "arxiv", + "source_id": "arxiv:2607.06001", + "pdf_url": "https://arxiv.org/pdf/2607.06001", + "primary_query": "llm-agent" + }, + { + "id": "2607.06000", + "title": "Context-to-Execution Integrity for LLM Agents", + "url": "https://arxiv.org/abs/2607.06000", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Igor Santos-Grueiro" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.06000", + "source": "arxiv", + "source_id": "arxiv:2607.06000", + "pdf_url": "https://arxiv.org/pdf/2607.06000", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.06413", + "title": "An Experimental Design Approach to Evaluating Agentic AI's Autonomous Model Discovery", + "url": "https://arxiv.org/abs/2607.06413", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Hao He", + "Xueying Liu", + "Chris J. Kuhlman", + "Xinwei Deng" + ], + "categories": [ + "stat.ME", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2607.06413", + "source": "arxiv", + "source_id": "arxiv:2607.06413", + "pdf_url": "https://arxiv.org/pdf/2607.06413", + "primary_query": "agentic-ai" + }, + { + "id": "2607.06411", + "title": "RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications", + "url": "https://arxiv.org/abs/2607.06411", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Evgeny Shilov" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2607.06411", + "source": "arxiv", + "source_id": "arxiv:2607.06411", + "pdf_url": "https://arxiv.org/pdf/2607.06411", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.06101", + "title": "Agents That Teach: Towards Designing Incidental Learning Back into AI-Assisted Software Development", + "url": "https://arxiv.org/abs/2607.06101", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Rohit Mehra", + "Samdyuti Suri", + "Prithviraj K Tagadinamani", + "Kapil Singi", + "Vikrant Kaulgud", + "Adam P. Burden" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CY", + "cs.HC" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.06101", + "source": "arxiv", + "source_id": "arxiv:2607.06101", + "pdf_url": "https://arxiv.org/pdf/2607.06101", + "primary_query": "coding-agent" + }, + { + "id": "2607.05659", + "title": "Agents with Feelings? Personality and Emotion in Multi-Agent Software Teams", + "url": "https://arxiv.org/abs/2607.05659", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Yunyan Ding", + "Thomas Zimmermann", + "Iftekhar Ahmed" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.05659", + "source": "arxiv", + "source_id": "arxiv:2607.05659", + "pdf_url": "https://arxiv.org/pdf/2607.05659", + "primary_query": "llm-agent" + }, + { + "id": "2607.04713", + "title": "RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents", + "url": "https://arxiv.org/abs/2607.04713", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Qiang Liu", + "Taian Guo", + "Ruizhi Qiao", + "Xing Sun" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent" + ], + "arxiv_id": "2607.04713", + "source": "arxiv", + "source_id": "arxiv:2607.04713", + "pdf_url": "https://arxiv.org/pdf/2607.04713", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.04569", + "title": "LLMs for Agentic Home Energy Management", + "url": "https://arxiv.org/abs/2607.04569", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Sokipriala Jonah" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling", + "llm-agent" + ], + "arxiv_id": "2607.04569", + "source": "arxiv", + "source_id": "arxiv:2607.04569", + "pdf_url": "https://arxiv.org/pdf/2607.04569", + "primary_query": "function-calling" + }, + { + "id": "2607.04617", + "title": "MRMS: A Multi-Resolution Memory Substrate for Long-Lived AI Agents", + "url": "https://arxiv.org/abs/2607.04617", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Jizhizi Li", + "Amy Shi-Nash" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.04617", + "source": "arxiv", + "source_id": "arxiv:2607.04617", + "pdf_url": "https://arxiv.org/pdf/2607.04617", + "primary_query": "ai-agent" + }, + { + "id": "2607.05363", + "title": "SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints", + "url": "https://arxiv.org/abs/2607.05363", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Dylan Zongmin Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "memory", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "tool-use" + ], + "arxiv_id": "2607.05363", + "source": "arxiv", + "source_id": "arxiv:2607.05363", + "pdf_url": "https://arxiv.org/pdf/2607.05363", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.05458", + "title": "Learning to Control LLM Agent Harnesses with Offline Reinforcement Learning", + "url": "https://arxiv.org/abs/2607.05458", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Haiwen Yi", + "Xinyuan Song" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.05458", + "source": "arxiv", + "source_id": "arxiv:2607.05458", + "pdf_url": "https://arxiv.org/pdf/2607.05458", + "primary_query": "llm-agent" + }, + { + "id": "2607.04219", + "title": "Agentic IoT: Architectures, Applications, and Challenges Toward the Internet of Agents", + "url": "https://arxiv.org/abs/2607.04219", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Rümeysa Hilal Sevinç", + "Bahaeddin Türkoğlu", + "İbrahim Kök" + ], + "categories": [ + "cs.AI", + "cs.MA", + "cs.NI" + ], + "topics": [ + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "tool-use" + ], + "arxiv_id": "2607.04219", + "source": "arxiv", + "source_id": "arxiv:2607.04219", + "pdf_url": "https://arxiv.org/pdf/2607.04219", + "primary_query": "ai-agent" + }, + { + "id": "2607.04212", + "title": "An Evaluation of Role-Based Multi-Agent Code Generation on Repository-Scale Problems", + "url": "https://arxiv.org/abs/2607.04212", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Benedetta Donato", + "Noah Hagar-Dent", + "Aaron Worsnop", + "Leonardo Mariani", + "Valerio Terragni" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.04212", + "source": "arxiv", + "source_id": "arxiv:2607.04212", + "pdf_url": "https://arxiv.org/pdf/2607.04212", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.04149", + "title": "Beyond Scene Priors: Fine-Grained Traffic Scene Reasoning with Benchmarking and Query-Guided Small-Object Focus", + "url": "https://arxiv.org/abs/2607.04149", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Waikit Xiu", + "Qiang Lu", + "Zian Wang", + "Xinjie Yang", + "Zhiwei Chen", + "Chen Sun", + "Xiying Li" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.04149", + "source": "arxiv", + "source_id": "arxiv:2607.04149", + "pdf_url": "https://arxiv.org/pdf/2607.04149", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.04034", + "title": "The \"I Don't Know\" Filter: Enhancing Agentic Reliability in Function Calling", + "url": "https://arxiv.org/abs/2607.04034", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Stefan Broecker", + "Mason del Rosario", + "Boris Selitser", + "Thomas Strohmer" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "function-calling" + ], + "arxiv_id": "2607.04034", + "source": "arxiv", + "source_id": "arxiv:2607.04034", + "pdf_url": "https://arxiv.org/pdf/2607.04034", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.03691", + "title": "Don't Blame the Large Language Model: How Scaffolding Evolution Shapes Coding Agent Quality", + "url": "https://arxiv.org/abs/2607.03691", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Oussama Ben Sghaier", + "Hao Li", + "Bram Adams", + "Ahmed E. Hassan" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03691", + "source": "arxiv", + "source_id": "arxiv:2607.03691", + "pdf_url": "https://arxiv.org/pdf/2607.03691", + "primary_query": "coding-agent" + }, + { + "id": "2607.02927", + "title": "VideoSearcher: Empowering Video Deep Research with Multi-Tool Agentic Reasoning via Reinforcement Learning", + "url": "https://arxiv.org/abs/2607.02927", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Zhenkun Gao", + "Yicheng Bao", + "Jinlong Peng", + "Xueheng Li", + "Theo Huang", + "Bangwei Liu", + "Kunquan Li", + "Zhenye Gan", + "Tao Hu", + "Chengjun Xie", + "Mingqian Yang", + "Xuanhua He", + "Zhizhong Zhang", + "Xin Tan", + "Chengjie Wang", + "Yuan Xie" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.02927", + "source": "arxiv", + "source_id": "arxiv:2607.02927", + "pdf_url": "https://arxiv.org/pdf/2607.02927", + "primary_query": "tool-use" + }, + { + "id": "2607.03525", + "title": "GameEngineBench: Evaluating Coding Agents on Real C++ Runtime Environments", + "url": "https://arxiv.org/abs/2607.03525", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Brian La", + "Sejoon Chang", + "Ben Kim", + "Junyoung Bae", + "Aamish Ahmad Beg", + "Sei Chang", + "Gonzalo Gonzalez-Pumariega" + ], + "categories": [ + "cs.SE", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "tool-use", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03525", + "source": "arxiv", + "source_id": "arxiv:2607.03525", + "pdf_url": "https://arxiv.org/pdf/2607.03525", + "primary_query": "coding-agent" + }, + { + "id": "2607.02689", + "title": "S-EMBER: A Large-Scale Benchmark for Streaming Egocentric Memory Retrieval", + "url": "https://arxiv.org/abs/2607.02689", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Xiaodong Wang", + "Xuanyi Zhao", + "Pedro Rodriguez", + "Devendra Singh Sachan", + "Barlas Oguz", + "Seungwhan Moon", + "Shang-Wen Li", + "Gargi Ghosh", + "Xin Dong", + "Wen-Tau Yih" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.02689", + "source": "arxiv", + "source_id": "arxiv:2607.02689", + "pdf_url": "https://arxiv.org/pdf/2607.02689", + "primary_query": "ai-agent" + }, + { + "id": "2607.01788", + "title": "KRCA: An Efficient Root Cause Analysis System in Hyper-Scale Microservice Systems via Agentic AI", + "url": "https://arxiv.org/abs/2607.01788", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Jiamin Jiang", + "Jingfei Feng", + "Yu Luo", + "Qingliang Zhang", + "Yongqian Su", + "Wenwei Gu", + "Shenglin Zhang", + "Tianyu Cui", + "Yao Wu", + "Jielong Huang", + "Nan Qi", + "Dan Pei" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.01788", + "source": "arxiv", + "source_id": "arxiv:2607.01788", + "pdf_url": "https://arxiv.org/pdf/2607.01788", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02716", + "title": "Evaluating Large Language Models for Decision-Making in Agent-Based Urban Mobility Simulations", + "url": "https://arxiv.org/abs/2607.02716", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Bruno Cascaes Alves", + "Míriam Blank Born", + "Ulisses Gilioli Francescatto Júnior", + "Felipe Moura Goulart", + "Letícia Brandão Caldas", + "Marilton Sanchotene de Aguiar" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.02716", + "source": "arxiv", + "source_id": "arxiv:2607.02716", + "pdf_url": "https://arxiv.org/pdf/2607.02716", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.02186", + "title": "UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development", + "url": "https://arxiv.org/abs/2607.02186", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Temitayo Olamilekan Ogunsusi", + "Lijun Qian", + "Xishuang Dong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.02186", + "source": "arxiv", + "source_id": "arxiv:2607.02186", + "pdf_url": "https://arxiv.org/pdf/2607.02186", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.01084", + "title": "Can Agents Generalize to the Open World? Unveiling the Fragility of Static Training in Tool Use", + "url": "https://arxiv.org/abs/2607.01084", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Song-Lin Lv", + "Weiming Wu", + "Rui Zhu", + "Zi-Jian Cheng", + "Lan-Zhe Guo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.01084", + "source": "arxiv", + "source_id": "arxiv:2607.01084", + "pdf_url": "https://arxiv.org/pdf/2607.01084", + "primary_query": "llm-agent" + }, + { + "id": "2607.02599", + "title": "AgentLTL: A Trace-Verification Framework for Measuring, Enforcing, and Training Procedural Compliance in Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2607.02599", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Laïla Elkoussy", + "Julien Perez" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.LO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.02599", + "source": "arxiv", + "source_id": "arxiv:2607.02599", + "pdf_url": "https://arxiv.org/pdf/2607.02599", + "primary_query": "llm-agent" + }, + { + "id": "2607.00555", + "title": "Rise From The Ashes: LLM-based Static Analysis for Deep Learning Framework Bugs", + "url": "https://arxiv.org/abs/2607.00555", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Shaoyu Yang", + "Haifeng Lin", + "Chunrong Fang", + "Xiang Chen", + "Wei Cheng", + "Jiawei Liu", + "Yiyu Zhang", + "Hongyu Liu", + "Zhenyu Chen" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2607.00555", + "source": "arxiv", + "source_id": "arxiv:2607.00555", + "pdf_url": "https://arxiv.org/pdf/2607.00555", + "primary_query": "agentic-ai" + }, + { + "id": "2607.00436", + "title": "PHREEQC-MCQ-200: A Diagnostic Benchmark for Tool-Augmented Scientific Simulator Agents", + "url": "https://arxiv.org/abs/2607.00436", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Ke Zhang", + "Sahchit Chundur", + "Mohammad Javad Qomi", + "Maziar Raissi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.00436", + "source": "arxiv", + "source_id": "arxiv:2607.00436", + "pdf_url": "https://arxiv.org/pdf/2607.00436", + "primary_query": "tool-use" + }, + { + "id": "2607.00502", + "title": "A Task-State Representation for Long-Horizon Mobile GUI Agents", + "url": "https://arxiv.org/abs/2607.00502", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Yujie Zheng", + "Zikang Liu", + "Xin Zhao", + "Ji-Rong Wen" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.00502", + "source": "arxiv", + "source_id": "arxiv:2607.00502", + "pdf_url": "https://arxiv.org/pdf/2607.00502", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.00440", + "title": "Minos: A Multi-Agent Collaborative Framework for Provenance-Based Backward Tracking", + "url": "https://arxiv.org/abs/2607.00440", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Jiahui Wang", + "Zhenyuan Li", + "Zhengkai Wang", + "Xiangmin Shen", + "Fan Zhang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00440", + "source": "arxiv", + "source_id": "arxiv:2607.00440", + "pdf_url": "https://arxiv.org/pdf/2607.00440", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00233", + "title": "From Signals to Structure: How Memory Architecture Drives Language Emergence in LLM Agents", + "url": "https://arxiv.org/abs/2607.00233", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yashar Talebirad", + "Eden Redman", + "Ali Parsaee", + "Osmar R. Zaiane" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.IT", + "cs.MA" + ], + "topics": [ + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00233", + "source": "arxiv", + "source_id": "arxiv:2607.00233", + "pdf_url": "https://arxiv.org/pdf/2607.00233", + "primary_query": "llm-agent" + }, + { + "id": "2606.32034", + "title": "QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2606.32034", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Sergio Hernández-Gutiérrez", + "Matteo Merler", + "Ilze Amanda Auzina", + "Joschka Strüber", + "Ameya Prabhu", + "Matthias Bethge" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.32034", + "source": "arxiv", + "source_id": "arxiv:2606.32034", + "pdf_url": "https://arxiv.org/pdf/2606.32034", + "primary_query": "llm-agent" + }, + { + "id": "2607.02579", + "title": "When Not to Write Memory: Governing False Promotion from Correlated Agent Traces", + "url": "https://arxiv.org/abs/2607.02579", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yijiashun Qi", + "Xiang Xu", + "Yuxuan Li" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "language-agent" + ], + "arxiv_id": "2607.02579", + "source": "arxiv", + "source_id": "arxiv:2607.02579", + "pdf_url": "https://arxiv.org/pdf/2607.02579", + "primary_query": "agent-memory" + }, + { + "id": "2606.31339", + "title": "Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems", + "url": "https://arxiv.org/abs/2606.31339", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Guoqin Tang", + "Qingxuan Jia", + "Yichen Tan", + "Zeyuan Huang", + "Ning Ji", + "Gang Chen" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.31339", + "source": "arxiv", + "source_id": "arxiv:2606.31339", + "pdf_url": "https://arxiv.org/pdf/2606.31339", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31980", + "title": "DigitalCoach: Communication and Grounding Gaps in Human and Agentic Computer Use Coaching", + "url": "https://arxiv.org/abs/2606.31980", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Meng Chen", + "Anya Ji", + "Tsung-Han Wu", + "Tobias Maringgele", + "David M. Chan", + "Alane Suhr", + "Amy Pavel" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.31980", + "source": "arxiv", + "source_id": "arxiv:2606.31980", + "pdf_url": "https://arxiv.org/pdf/2606.31980", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.31134", + "title": "Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics", + "url": "https://arxiv.org/abs/2606.31134", + "published": "2026-06-30", + "updated": "2026-07-01", + "authors": [ + "Arshia Soltani Moakhar", + "Iman Gholami", + "Max Springer", + "Mahdi JafariRaviz", + "MohammadTaghi Hajiaghayi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31134", + "source": "arxiv", + "source_id": "arxiv:2606.31134", + "pdf_url": "https://arxiv.org/pdf/2606.31134", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31200", + "title": "Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping", + "url": "https://arxiv.org/abs/2606.31200", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Tao Chen", + "Lizheng Liu", + "Jiaxu Wang", + "Ziyue Jiang", + "Ruiqi Tian", + "JiGuang Huo", + "Zhongxue Gan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.31200", + "source": "arxiv", + "source_id": "arxiv:2606.31200", + "pdf_url": "https://arxiv.org/pdf/2606.31200", + "primary_query": "rag-agent" + }, + { + "id": "2606.30251", + "title": "TACO: Tool-Augmented Credit Optimization for Agentic Tool Use", + "url": "https://arxiv.org/abs/2606.30251", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Mingkuan Feng", + "Jinyang Wu", + "Hao Gu", + "Fangrui Lv", + "Ruihan Jin", + "Chuyuan Zhang", + "Zhengqi Wen", + "Jianhua Tao" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.30251", + "source": "arxiv", + "source_id": "arxiv:2606.30251", + "pdf_url": "https://arxiv.org/pdf/2606.30251", + "primary_query": "tool-use" + }, + { + "id": "2606.30294", + "title": "Rehearsed Multi-Agent Live Product Demonstrations with Real-Time Voice Question Answering", + "url": "https://arxiv.org/abs/2606.30294", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Rahul Khedar", + "Mayank Malhotra", + "Avinash Karn", + "Mouli V", + "Prakhar Mehrotra" + ], + "categories": [ + "cs.AI", + "cs.HC", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.30294", + "source": "arxiv", + "source_id": "arxiv:2606.30294", + "pdf_url": "https://arxiv.org/pdf/2606.30294", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.29354", + "title": "When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning", + "url": "https://arxiv.org/abs/2606.29354", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Zhengqi Pei", + "Qingming Huang", + "Shuhui Wang" + ], + "categories": [ + "cs.AI", + "cs.NE" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.29354", + "source": "arxiv", + "source_id": "arxiv:2606.29354", + "pdf_url": "https://arxiv.org/pdf/2606.29354", + "primary_query": "llm-agent" + }, + { + "id": "2606.29315", + "title": "Hierarchical Experimentalist Agents", + "url": "https://arxiv.org/abs/2606.29315", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Abhranil Chandra", + "Sankaran Vaidyanathan", + "Utsav Dhanuka", + "Varun Gandhi", + "Scott Niekum" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29315", + "source": "arxiv", + "source_id": "arxiv:2606.29315", + "pdf_url": "https://arxiv.org/pdf/2606.29315", + "primary_query": "llm-agent" + }, + { + "id": "2606.29445", + "title": "Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction", + "url": "https://arxiv.org/abs/2606.29445", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Sunqi Fan", + "Qingle Liu", + "Runqi Yin", + "Meng-Hao Guo", + "Shuojin Yang" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.29445", + "source": "arxiv", + "source_id": "arxiv:2606.29445", + "pdf_url": "https://arxiv.org/pdf/2606.29445", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.28841", + "title": "LAMP: Lean-based Agentic framework with MCP and Proof Repair", + "url": "https://arxiv.org/abs/2606.28841", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Santhana Srinivasan R", + "Maithilee Patawar" + ], + "categories": [ + "cs.LO", + "cs.AI", + "cs.CL" + ], + "topics": [ + "coding-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28841", + "source": "arxiv", + "source_id": "arxiv:2606.28841", + "pdf_url": "https://arxiv.org/pdf/2606.28841", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28450", + "title": "LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity", + "url": "https://arxiv.org/abs/2606.28450", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Yiwei Xu", + "Yong Zhuang", + "Xuanming Liu", + "Tian Zhang", + "Bowen Xiao", + "Xiaoyang Xu", + "Delong Jiang", + "Juan Wang", + "Hongxin Hu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.28450", + "source": "arxiv", + "source_id": "arxiv:2606.28450", + "pdf_url": "https://arxiv.org/pdf/2606.28450", + "primary_query": "agent-safety" + }, + { + "id": "2606.28480", + "title": "TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents", + "url": "https://arxiv.org/abs/2606.28480", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Shoufa Chen", + "Luyuan Wang", + "Xuan Yang", + "Zhiheng Liu", + "Yuren Cong", + "Yuanfeng Ji", + "Feiyan Zhou", + "Xiaohui Zhang", + "Fanny Yang", + "Belinda Zeng" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "reasoning", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.28480", + "source": "arxiv", + "source_id": "arxiv:2606.28480", + "pdf_url": "https://arxiv.org/pdf/2606.28480", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.27350", + "title": "CHIA: An open-source framework for principled, agentic AI-driven hardware/software co-design research", + "url": "https://arxiv.org/abs/2606.27350", + "published": "2026-06-25", + "updated": "2026-06-27", + "authors": [ + "Angela Cui", + "Ferran Hermida-Rivera", + "Jack Toubes", + "Raghav Gupta", + "Jim Fang", + "Chengyi Lux Zhang", + "Ella Schwarz", + "Junha Kim", + "Yakun Sophia Shao", + "Borivoje Nikolic", + "Christopher W. Fletcher", + "Sagar Karandikar" + ], + "categories": [ + "cs.AR" + ], + "topics": [ + "agent-safety", + "coding-agent", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2606.27350", + "source": "arxiv", + "source_id": "arxiv:2606.27350", + "pdf_url": "https://arxiv.org/pdf/2606.27350", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26721", + "title": "Knowledge-Based Pull Requests: A Trusted Workflow for Agent-Mediated Knowledge Collaboration", + "url": "https://arxiv.org/abs/2606.26721", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Xinyu Zhang", + "Weiwei Sun" + ], + "categories": [ + "cs.SE", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.26721", + "source": "arxiv", + "source_id": "arxiv:2606.26721", + "pdf_url": "https://arxiv.org/pdf/2606.26721", + "primary_query": "coding-agent" + }, + { + "id": "2606.27330", + "title": "Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning", + "url": "https://arxiv.org/abs/2606.27330", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Tianyi Men", + "Zhuoran Jin", + "Pengfei Cao", + "Yubo Chen", + "Kang Liu", + "Jun Zhao" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.CV", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.27330", + "source": "arxiv", + "source_id": "arxiv:2606.27330", + "pdf_url": "https://arxiv.org/pdf/2606.27330", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.27009", + "title": "Semantic Early-Stopping for Iterative LLM Agent Loops", + "url": "https://arxiv.org/abs/2606.27009", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Sahil Shrivastava" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.27009", + "source": "arxiv", + "source_id": "arxiv:2606.27009", + "pdf_url": "https://arxiv.org/pdf/2606.27009", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.27483", + "title": "Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning", + "url": "https://arxiv.org/abs/2606.27483", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Xuan Zhang", + "Zhijian Zhou", + "Lingfeng Qiao", + "Yulei Qin", + "Ke Li", + "Xing Sun", + "Xiaoyu Tan", + "Chao Qu", + "Yuan Qi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.27483", + "source": "arxiv", + "source_id": "arxiv:2606.27483", + "pdf_url": "https://arxiv.org/pdf/2606.27483", + "primary_query": "planning-agent" + }, + { + "id": "2606.26216", + "title": "CyberChainBench: Can AI Agents Secure Smart Contracts Against Real-World On-Chain Vulnerabilities?", + "url": "https://arxiv.org/abs/2606.26216", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Jintao Huang", + "Fengqing Jiang", + "Radha Poovendran", + "Zhiqiang Lin" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "ai-agent" + ], + "arxiv_id": "2606.26216", + "source": "arxiv", + "source_id": "arxiv:2606.26216", + "pdf_url": "https://arxiv.org/pdf/2606.26216", + "primary_query": "agent-safety" + }, + { + "id": "2606.26203", + "title": "Agentic Analysis for Agentic Infrastructure: An LLM-Powered Pipeline for Comparative Governance of DAO and Corporate AI Protocols", + "url": "https://arxiv.org/abs/2606.26203", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yutian Wang", + "Luyao Zhang" + ], + "categories": [ + "cs.AI", + "cs.CY", + "cs.MA", + "cs.SI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent" + ], + "arxiv_id": "2606.26203", + "source": "arxiv", + "source_id": "arxiv:2606.26203", + "pdf_url": "https://arxiv.org/pdf/2606.26203", + "primary_query": "agentic-ai" + }, + { + "id": "2606.25760", + "title": "Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets", + "url": "https://arxiv.org/abs/2606.25760", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Divake Kumar", + "Sina Tayebati", + "Devashri Naik", + "Amanda Sofie Rios", + "Nilesh Ahuja", + "Omesh Tickoo", + "Ranganath Krishnan", + "Amit Ranjan Trivedi" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "web-gui-agent" + ], + "arxiv_id": "2606.25760", + "source": "arxiv", + "source_id": "arxiv:2606.25760", + "pdf_url": "https://arxiv.org/pdf/2606.25760", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.25651", + "title": "MedGuards: Multi-Agent System for Reliable Medical Error Detection and Correction", + "url": "https://arxiv.org/abs/2606.25651", + "published": "2026-06-24", + "updated": "2026-06-26", + "authors": [ + "Congbo Ma", + "Hu Wang", + "Yichun Zhang", + "Farah E. Shamout" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25651", + "source": "arxiv", + "source_id": "arxiv:2606.25651", + "pdf_url": "https://arxiv.org/pdf/2606.25651", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24595", + "title": "MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery", + "url": "https://arxiv.org/abs/2606.24595", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Enze Ma", + "Yufan Zhou", + "Wei-Chieh Huang", + "Jie Yang", + "Huanhuan Ma", + "Zixuan Wang", + "Chengze Li", + "Chunyu Miao", + "Philip S. Yu", + "Zhen Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.24595", + "source": "arxiv", + "source_id": "arxiv:2606.24595", + "pdf_url": "https://arxiv.org/pdf/2606.24595", + "primary_query": "agent-memory" + }, + { + "id": "2606.24453", + "title": "Bayesian control for coding agents", + "url": "https://arxiv.org/abs/2606.24453", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Theodore Papamarkou", + "Vladislav Smirnov", + "Viktor Mazanov", + "Artem Vazhentsev", + "Preslav Nakov", + "Timothy Baldwin", + "Artem Shelmanov" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "tool-use" + ], + "arxiv_id": "2606.24453", + "source": "arxiv", + "source_id": "arxiv:2606.24453", + "pdf_url": "https://arxiv.org/pdf/2606.24453", + "primary_query": "coding-agent" + }, + { + "id": "2606.24525", + "title": "VisCritic: Visual State Comparison as Process Reward for GUI Agents", + "url": "https://arxiv.org/abs/2606.24525", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Jiachen Qian" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.24525", + "source": "arxiv", + "source_id": "arxiv:2606.24525", + "pdf_url": "https://arxiv.org/pdf/2606.24525", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.24689", + "title": "Automated Summarization of Software Documents: An LLM-based Multi-Agent Approach", + "url": "https://arxiv.org/abs/2606.24689", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Duc S. H. Nguyen", + "Minh T. Nguyen", + "Phuong T. Nguyen", + "Juri Di Rocco", + "Davide Di Ruscio" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24689", + "source": "arxiv", + "source_id": "arxiv:2606.24689", + "pdf_url": "https://arxiv.org/pdf/2606.24689", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24623", + "title": "Privacy-Preserving RAG via Multi-Agent Semantic Rewriting: Achieving Confidentiality Without Compromising Contextual Fidelity", + "url": "https://arxiv.org/abs/2606.24623", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yuanhe Zhao", + "Tianyu Zhang", + "Huafei Xing", + "Derek F. Wong", + "Jianbin Li", + "Tao Fang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "rag-agent" + ], + "arxiv_id": "2606.24623", + "source": "arxiv", + "source_id": "arxiv:2606.24623", + "pdf_url": "https://arxiv.org/pdf/2606.24623", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.23664", + "title": "MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems?", + "url": "https://arxiv.org/abs/2606.23664", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Juyang Bai", + "Laixi Shi" + ], + "categories": [ + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.23664", + "source": "arxiv", + "source_id": "arxiv:2606.23664", + "pdf_url": "https://arxiv.org/pdf/2606.23664", + "primary_query": "agentic-ai" + }, + { + "id": "2606.23032", + "title": "IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO", + "url": "https://arxiv.org/abs/2606.23032", + "published": "2026-06-22", + "updated": "2026-06-30", + "authors": [ + "Mostapha Benhenda" + ], + "categories": [ + "cs.AI", + "q-fin.GN" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.23032", + "source": "arxiv", + "source_id": "arxiv:2606.23032", + "pdf_url": "https://arxiv.org/pdf/2606.23032", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.22741", + "title": "GRADE: Graph Representation of LLM Agent Dependency and Execution", + "url": "https://arxiv.org/abs/2606.22741", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Yue Zhao" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "coding-agent", + "multi-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.22741", + "source": "arxiv", + "source_id": "arxiv:2606.22741", + "pdf_url": "https://arxiv.org/pdf/2606.22741", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.23752", + "title": "ESAA-Conversational: An Event-Sourced Memory Layer for Continuity, Handoff, and Curation Across Heterogeneous LLM Coding Agents", + "url": "https://arxiv.org/abs/2606.23752", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Elzo Brito dos Santos Filho" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "coding-agent", + "memory", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.23752", + "source": "arxiv", + "source_id": "arxiv:2606.23752", + "pdf_url": "https://arxiv.org/pdf/2606.23752", + "primary_query": "coding-agent" + }, + { + "id": "2606.23327", + "title": "VideoAgent: All-in-One Framework for Video Understanding and Editing", + "url": "https://arxiv.org/abs/2606.23327", + "published": "2026-06-22", + "updated": "2026-07-03", + "authors": [ + "Hengji Zhou", + "Lingxuan Huang", + "Jian Wang", + "Bing Zhou", + "Si Wu", + "Lianghao Xia", + "Chao Huang" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.23327", + "source": "arxiv", + "source_id": "arxiv:2606.23327", + "pdf_url": "https://arxiv.org/pdf/2606.23327", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22110", + "title": "TraceView: Interactive Visualization of Agentic Program Repair Trajectories", + "url": "https://arxiv.org/abs/2606.22110", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Amirali Sajadi", + "Tu Nguyen", + "Kimmie Huynh", + "Esteban Parra", + "Preetha Chatterjee" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.22110", + "source": "arxiv", + "source_id": "arxiv:2606.22110", + "pdf_url": "https://arxiv.org/pdf/2606.22110", + "primary_query": "tool-use" + }, + { + "id": "2606.22082", + "title": "CodeTeam: An LLM-Powered Multi-Agent Framework for Repository-Level Code Generation", + "url": "https://arxiv.org/abs/2606.22082", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Yifei Wang", + "Ruiyin Li", + "Peng Liang", + "Qiong Feng", + "Zengyang Li", + "Mojtaba Shahin", + "Arif Ali Khan" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.22082", + "source": "arxiv", + "source_id": "arxiv:2606.22082", + "pdf_url": "https://arxiv.org/pdf/2606.22082", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.21963", + "title": "Holmes: Multimodal Agentic Diagnosis for Mixed-Language Mobile Crashes at Industrial Scale", + "url": "https://arxiv.org/abs/2606.21963", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Jia Li", + "Wenyuan Ma", + "Ting Peng", + "Haibin Zheng", + "Yuetang Deng" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "rag", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.21963", + "source": "arxiv", + "source_id": "arxiv:2606.21963", + "pdf_url": "https://arxiv.org/pdf/2606.21963", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.21445", + "title": "AutoRAS: Learning Robust Agentic Systems with Primitive Representations", + "url": "https://arxiv.org/abs/2606.21445", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Yang Yue", + "Xuancheng Zhu", + "Yuyang Ma", + "Guoshun Nan", + "Zihan Dou", + "Jingru Shan", + "Congyu Guo", + "Ji Zhang", + "Hua Wang", + "Jingfeng Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.21445", + "source": "arxiv", + "source_id": "arxiv:2606.21445", + "pdf_url": "https://arxiv.org/pdf/2606.21445", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20954", + "title": "Learning What Not to Forget: Long-Horizon Agent Memory from a Few Kilobytes of Learning", + "url": "https://arxiv.org/abs/2606.20954", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Nusrat Jahan Lia", + "Aritra Mazumder" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.20954", + "source": "arxiv", + "source_id": "arxiv:2606.20954", + "pdf_url": "https://arxiv.org/pdf/2606.20954", + "primary_query": "agent-memory" + }, + { + "id": "2606.19980", + "title": "ENPIRE: Agentic Robot Policy Self-Improvement in the Real World", + "url": "https://arxiv.org/abs/2606.19980", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Wenli Xiao", + "Jia Xie", + "Tonghe Zhang", + "Haotian Lin", + "Letian \"Max\" Fu", + "Haoru Xue", + "Jalen Lu", + "Yi Yang", + "Cunxi Dai", + "Zi Wang", + "Jimmy Wu", + "Guanzhi Wang", + "S. Shankar Sastry", + "Ken Goldberg", + "Linxi \"Jim\" Fan", + "Yuke Zhu", + "Guanya Shi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "tool-use" + ], + "arxiv_id": "2606.19980", + "source": "arxiv", + "source_id": "arxiv:2606.19980", + "pdf_url": "https://arxiv.org/pdf/2606.19980", + "primary_query": "coding-agent" + }, + { + "id": "2606.19245", + "title": "TxBench-PP: Analyzing AI Agent Performance on Small-Molecule Preclinical Pharmacology", + "url": "https://arxiv.org/abs/2606.19245", + "published": "2026-06-17", + "updated": "2026-06-18", + "authors": [ + "Hannah Le", + "Ramesh Ramasamy", + "Alex Urrutia", + "Mahsa Yazdani", + "Tim Proctor", + "Kenny Workman" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.19245", + "source": "arxiv", + "source_id": "arxiv:2606.19245", + "pdf_url": "https://arxiv.org/pdf/2606.19245", + "primary_query": "ai-agent" + }, + { + "id": "2606.18619", + "title": "Code-Augur: Agentic Vulnerability Detection via Specification Inference", + "url": "https://arxiv.org/abs/2606.18619", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Zhengxiong Luo", + "Mehtab Zafar", + "Dylan Wolff", + "Abhik Roychoudhury" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-safety", + "computer-use", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.18619", + "source": "arxiv", + "source_id": "arxiv:2606.18619", + "pdf_url": "https://arxiv.org/pdf/2606.18619", + "primary_query": "ai-agent" + }, + { + "id": "2606.19464", + "title": "Deontic Policies for Runtime Governance of Agentic AI Systems", + "url": "https://arxiv.org/abs/2606.19464", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Anupam Joshi", + "Tim Finin", + "Karuna Pande Joshi", + "Lalana Kagal" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.19464", + "source": "arxiv", + "source_id": "arxiv:2606.19464", + "pdf_url": "https://arxiv.org/pdf/2606.19464", + "primary_query": "agentic-ai" + }, + { + "id": "2606.18502", + "title": "Towards Scalable Customization and Deployment of Multi-Agent Systems for Enterprise Applications", + "url": "https://arxiv.org/abs/2606.18502", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Paresh Dashore", + "Shreyas Kulkarni", + "Uttam Gurram", + "Nadia Bathaee", + "Kartik Balasubramaniam", + "Genta Indra Winata", + "Sambit Sahu", + "Shi-Xiong Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "coding-agent", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.18502", + "source": "arxiv", + "source_id": "arxiv:2606.18502", + "pdf_url": "https://arxiv.org/pdf/2606.18502", + "primary_query": "agentic-ai" + }, + { + "id": "2606.18037", + "title": "ProvenanceGuard: Source-Aware Factuality Verification for MCP-Based LLM Agents", + "url": "https://arxiv.org/abs/2606.18037", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Ander Alvarez", + "Santhiya Rajan", + "Samuel Mugel", + "Román Orús" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.18037", + "source": "arxiv", + "source_id": "arxiv:2606.18037", + "pdf_url": "https://arxiv.org/pdf/2606.18037", + "primary_query": "tool-use" + }, + { + "id": "2606.17573", + "title": "Cordon: Semantic Transactions for Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.17573", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Zheng Chen", + "Hanqing Liu", + "Duling Xu", + "Dong Dong", + "Jialin Li", + "Bangzheng Pu", + "Jidong Zhai" + ], + "categories": [ + "cs.OS", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.17573", + "source": "arxiv", + "source_id": "arxiv:2606.17573", + "pdf_url": "https://arxiv.org/pdf/2606.17573", + "primary_query": "tool-use" + }, + { + "id": "2606.18051", + "title": "Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose", + "url": "https://arxiv.org/abs/2606.18051", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Xueping Gao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.18051", + "source": "arxiv", + "source_id": "arxiv:2606.18051", + "pdf_url": "https://arxiv.org/pdf/2606.18051", + "primary_query": "planning-agent" + }, + { + "id": "2606.16295", + "title": "VisualClaw: A Real-Time, Personalized Agent for the Physical World", + "url": "https://arxiv.org/abs/2606.16295", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Haoqin Tu", + "Jianwen Chen", + "Zijun Wang", + "Siwei Han", + "Juncheng Wu", + "Hardy Chen", + "Haonian Ji", + "Kaiwen Xiong", + "Jiaqi Liu", + "Peng Xia", + "Jieru Mei", + "Hongliang Fei", + "Jason Eshraghian", + "Zeyu Zheng", + "Yuyin Zhou", + "Huaxiu Yao", + "Cihang Xie" + ], + "categories": [ + "cs.CV", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "tool-use", + "web-gui-agent" + ], + "arxiv_id": "2606.16295", + "source": "arxiv", + "source_id": "arxiv:2606.16295", + "pdf_url": "https://arxiv.org/pdf/2606.16295", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.16534", + "title": "Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers", + "url": "https://arxiv.org/abs/2606.16534", + "published": "2026-06-15", + "updated": "2026-06-16", + "authors": [ + "Linus Bantel", + "Anna-Lena Roth", + "Jonas Posner", + "Dirk Pflüger" + ], + "categories": [ + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.16534", + "source": "arxiv", + "source_id": "arxiv:2606.16534", + "pdf_url": "https://arxiv.org/pdf/2606.16534", + "primary_query": "tool-use" + }, + { + "id": "2606.15376", + "title": "CoAgent: Concurrency Control for Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.15376", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Hongtao Lyu", + "Dingyan Zhang", + "Mingyu Wu", + "Xingda Wei", + "Haibo Chen" + ], + "categories": [ + "cs.DC", + "cs.AI", + "cs.MA" + ], + "topics": [ + "coding-agent", + "multi-agent", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.15376", + "source": "arxiv", + "source_id": "arxiv:2606.15376", + "pdf_url": "https://arxiv.org/pdf/2606.15376", + "primary_query": "planning-agent" + }, + { + "id": "2606.14517", + "title": "From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails", + "url": "https://arxiv.org/abs/2606.14517", + "published": "2026-06-12", + "updated": "2026-06-16", + "authors": [ + "Yuguang Zhou", + "Xunguang Wang", + "Pingchuan Ma", + "Zhantong Xue", + "Zhaoyu Wang", + "Shuai Wang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.14517", + "source": "arxiv", + "source_id": "arxiv:2606.14517", + "pdf_url": "https://arxiv.org/pdf/2606.14517", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.14470", + "title": "GitOfThoughts: Version-Controlled Reasoning and Agent Memory You Can Replay, Diff, and Merge", + "url": "https://arxiv.org/abs/2606.14470", + "published": "2026-06-12", + "updated": "2026-06-22", + "authors": [ + "Pavan C Shekar", + "Abhishek H S", + "Aswanth Krishnan" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.14470", + "source": "arxiv", + "source_id": "arxiv:2606.14470", + "pdf_url": "https://arxiv.org/pdf/2606.14470", + "primary_query": "agent-memory" + }, + { + "id": "2606.14106", + "title": "Naive Visual Memory is Not Enough: A Failure-Mode Study of GUI Agents", + "url": "https://arxiv.org/abs/2606.14106", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Seoyoung Choi", + "Minseok Ko", + "Hyunseok Lee", + "Kunwoong Kim", + "Woomin Song", + "Chanseok Jeon", + "Jinwoo Shin" + ], + "categories": [ + "cs.MA", + "cs.CV" + ], + "topics": [ + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.14106", + "source": "arxiv", + "source_id": "arxiv:2606.14106", + "pdf_url": "https://arxiv.org/pdf/2606.14106", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13608", + "title": "AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility", + "url": "https://arxiv.org/abs/2606.13608", + "published": "2026-06-11", + "updated": "2026-06-14", + "authors": [ + "Xiaoyuan Liu", + "Jianhong Tu", + "Yuqi Chen", + "Siyuan Xie", + "Sihan Ren", + "Tianneng Shi", + "Gal Gantar", + "Evan Sandoval", + "Donghyun Lee", + "Daniel Miao", + "Peter J. Gilbert", + "Nick Hynes", + "Mauro Staver", + "Warren He", + "David Marn", + "Andrew Low", + "Xi Zhang", + "Elron Bandel", + "Michal Shmueli-Scheuer", + "Siva Reddy", + "Alexandre Drouin", + "Alexandre Lacoste", + "Ramayya Krishnan", + "Elham Tabassi", + "Yu Su", + "Victor Barres", + "Chenguang Wang", + "Wenbo Guo", + "Dawn Song" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.13608", + "source": "arxiv", + "source_id": "arxiv:2606.13608", + "pdf_url": "https://arxiv.org/pdf/2606.13608", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.13177", + "title": "MemRefine: LLM-Guided Compression for Long-Term Agent Memory", + "url": "https://arxiv.org/abs/2606.13177", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Minjae Kim", + "Jinheon Baek", + "Soyeong Jeong", + "Sung Ju Hwang" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.13177", + "source": "arxiv", + "source_id": "arxiv:2606.13177", + "pdf_url": "https://arxiv.org/pdf/2606.13177", + "primary_query": "agent-memory" + }, + { + "id": "2606.13148", + "title": "TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?", + "url": "https://arxiv.org/abs/2606.13148", + "published": "2026-06-11", + "updated": "2026-07-01", + "authors": [ + "Dat Tien Nguyen", + "Thao Nguyen", + "Fadillah Adamsyah Maani", + "Huy M. Le", + "Muhammad Umer Sheikh", + "Numan Saeed", + "Muhammad Haris Khan", + "Salman Khan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.13148", + "source": "arxiv", + "source_id": "arxiv:2606.13148", + "pdf_url": "https://arxiv.org/pdf/2606.13148", + "primary_query": "tool-use" + }, + { + "id": "2606.28360", + "title": "Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance", + "url": "https://arxiv.org/abs/2606.28360", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Ben Torsion", + "Jun Zhou" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.28360", + "source": "arxiv", + "source_id": "arxiv:2606.28360", + "pdf_url": "https://arxiv.org/pdf/2606.28360", + "primary_query": "rag-agent" + }, + { + "id": "2606.12344", + "title": "Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks", + "url": "https://arxiv.org/abs/2606.12344", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Mengyu Zheng", + "Kai Han", + "Boxun Li", + "Haiyang Xu", + "Yuchuan Tian", + "Wei He", + "Hang Zhou", + "Jianyuan Guo", + "Hailin Hu", + "Lin Ma", + "Chao Xu", + "Guohao Dai", + "Lixue Xia", + "Yunchao Wei", + "Yunhe Wang", + "Yu Wang" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.12344", + "source": "arxiv", + "source_id": "arxiv:2606.12344", + "pdf_url": "https://arxiv.org/pdf/2606.12344", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.11702", + "title": "MedCTA: A Benchmark for Clinical Tool Agents", + "url": "https://arxiv.org/abs/2606.11702", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Tajamul Ashraf", + "Hyewon Jeong", + "Fida Mohammad Thoker", + "Bernard Ghanem" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.11702", + "source": "arxiv", + "source_id": "arxiv:2606.11702", + "pdf_url": "https://arxiv.org/pdf/2606.11702", + "primary_query": "tool-use" + }, + { + "id": "2606.11119", + "title": "TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.11119", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Heming Zou", + "Qi Wang", + "Yun Qu", + "Yuhang Jiang", + "Lizhou Cai", + "Yixiu Mao", + "Ru Peng", + "Xin Xu", + "Weijie Liu", + "Kai Yang", + "Saiyong Yang", + "Xiangyang Ji" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.11119", + "source": "arxiv", + "source_id": "arxiv:2606.11119", + "pdf_url": "https://arxiv.org/pdf/2606.11119", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.10921", + "title": "Trace Only What You Need: Structure-Aware On-Demand Hypergraph Memory for Long-Document Question Answering", + "url": "https://arxiv.org/abs/2606.10921", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Xiangjun Zai", + "Xingyu Tan", + "Chen Chen", + "Xiaoyang Wang", + "Wenjie Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.10921", + "source": "arxiv", + "source_id": "arxiv:2606.10921", + "pdf_url": "https://arxiv.org/pdf/2606.10921", + "primary_query": "rag-agent" + }, + { + "id": "2606.10316", + "title": "TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning", + "url": "https://arxiv.org/abs/2606.10316", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Mingyue Cheng", + "Shuo Yu", + "Daoyu Wang", + "Qingchuan Li", + "Xiaoyu Tao", + "Qingyang Mao", + "Yitong Zhou", + "Qi Liu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.10316", + "source": "arxiv", + "source_id": "arxiv:2606.10316", + "pdf_url": "https://arxiv.org/pdf/2606.10316", + "primary_query": "planning-agent" + }, + { + "id": "2606.09037", + "title": "A Multi-Agent System for IPMSM Design Optimization via an FEA-AI Hybrid Approach", + "url": "https://arxiv.org/abs/2606.09037", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Jinseong Han", + "Sunwoong Yang", + "Namwoo Kang" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.09037", + "source": "arxiv", + "source_id": "arxiv:2606.09037", + "pdf_url": "https://arxiv.org/pdf/2606.09037", + "primary_query": "rag-agent" + }, + { + "id": "2606.09738", + "title": "HDSL: A Hierarchical Domain-Specific Language for Structured 3D Indoor Scene Generation and Localized Editing with LLM Agents", + "url": "https://arxiv.org/abs/2606.09738", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Letian Li", + "Chao Shen", + "Shuzhao Xie", + "Chenghao Gu", + "ZhengXiao He", + "Yu Meng", + "Xin Yang", + "Wenyuan Jiang", + "Zhi Wang" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.09738", + "source": "arxiv", + "source_id": "arxiv:2606.09738", + "pdf_url": "https://arxiv.org/pdf/2606.09738", + "primary_query": "planning-agent" + }, + { + "id": "2606.10209", + "title": "Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.10209", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Abhilasha Lodha", + "Mahsa Pahlavikhah Varnosfaderani", + "Abir Chakraborty", + "Abhinav Mithal" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.10209", + "source": "arxiv", + "source_id": "arxiv:2606.10209", + "pdf_url": "https://arxiv.org/pdf/2606.10209", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.05711", + "title": "Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems", + "url": "https://arxiv.org/abs/2606.05711", + "published": "2026-06-04", + "updated": "2026-06-05", + "authors": [ + "Yingzhuo Liu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.05711", + "source": "arxiv", + "source_id": "arxiv:2606.05711", + "pdf_url": "https://arxiv.org/pdf/2606.05711", + "primary_query": "language-agent" + }, + { + "id": "2606.06090", + "title": "Beyond Semantic Organization: Memory as Execution State Management for Long-Horizon Agents", + "url": "https://arxiv.org/abs/2606.06090", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Yaoqi Chen", + "Haibin Lai", + "Yuru Feng", + "Chuyu Han", + "Qianxi Zhang", + "Baotong Lu", + "Menghao Li", + "Xinjiang Wang", + "Zhirui Wang", + "Shusen Xu", + "Zengzhong Li", + "Zewen Jin", + "Hao Wu", + "Cheng Li", + "Qi Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2606.06090", + "source": "arxiv", + "source_id": "arxiv:2606.06090", + "pdf_url": "https://arxiv.org/pdf/2606.06090", + "primary_query": "agent-memory" + }, + { + "id": "2606.06388", + "title": "Humans' ALMANAC: A Human Collaboration Dataset of Action-Level Mental Model Annotations for Agent Collaboration", + "url": "https://arxiv.org/abs/2606.06388", + "published": "2026-06-04", + "updated": "2026-06-06", + "authors": [ + "Jiaju Chen", + "Yuxuan Lu", + "Jiayi Su", + "Chaoran Chen", + "Songlin Xiao", + "Zheng Zhang", + "Yun Wang", + "Yunyao Li", + "Jian Zhao", + "Tongshuang Wu", + "Toby Jia-Jun Li", + "Dakuo Wang", + "Bingsheng Yao" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.06388", + "source": "arxiv", + "source_id": "arxiv:2606.06388", + "pdf_url": "https://arxiv.org/pdf/2606.06388", + "primary_query": "planning-agent" + }, + { + "id": "2606.05241", + "title": "Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation", + "url": "https://arxiv.org/abs/2606.05241", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Yongjie Wang", + "Xinyue Zhang", + "Kunhong Yao", + "Zhiwei Zeng", + "Kaisong Song", + "Jun Lin", + "Zhiqi Shen" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.05241", + "source": "arxiv", + "source_id": "arxiv:2606.05241", + "pdf_url": "https://arxiv.org/pdf/2606.05241", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.03197", + "title": "MemTrain: Self-Supervised Context Memory Training", + "url": "https://arxiv.org/abs/2606.03197", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Ziheng Li", + "Xingrun Xing", + "Haoqing Wang", + "Zhi-Hong Deng", + "Yehui Tang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.03197", + "source": "arxiv", + "source_id": "arxiv:2606.03197", + "pdf_url": "https://arxiv.org/pdf/2606.03197", + "primary_query": "agent-memory" + }, + { + "id": "2606.02497", + "title": "Bridging the Last Mile of Time Series Forecasting with LLM Agents", + "url": "https://arxiv.org/abs/2606.02497", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Yuhua Liao", + "Zetian Wang", + "Qiangqiang Nie", + "Zhenhua Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.02497", + "source": "arxiv", + "source_id": "arxiv:2606.02497", + "pdf_url": "https://arxiv.org/pdf/2606.02497", + "primary_query": "planning-agent" + }, + { + "id": "2606.01222", + "title": "RAG-driven Multi-Agent LLM Framework with Task Decomposition for Beyond 5G Auto-Configuration", + "url": "https://arxiv.org/abs/2606.01222", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "İrşat Emin Sarıdaş", + "Onur Salan", + "Ali Görçin", + "Ibrahim Hokelek", + "Hakan Ali Çırpan" + ], + "categories": [ + "eess.SP" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.01222", + "source": "arxiv", + "source_id": "arxiv:2606.01222", + "pdf_url": "https://arxiv.org/pdf/2606.01222", + "primary_query": "rag-agent" + }, + { + "id": "2606.00619", + "title": "MemPro: Agentic Memory Systems as Evolvable Programs", + "url": "https://arxiv.org/abs/2606.00619", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Qingshan Liu", + "Guoqing Wang", + "Wen Wu", + "Jingqi Huang", + "Xinqi Tao", + "Dejia Song", + "Jie Zhou", + "Liang He" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.00619", + "source": "arxiv", + "source_id": "arxiv:2606.00619", + "pdf_url": "https://arxiv.org/pdf/2606.00619", + "primary_query": "agent-memory" + }, + { + "id": "2606.00922", + "title": "A Machine-to-Machine Knowledge-Guided LLM Agent for Generalizable Radiotherapy Treatment Planning", + "url": "https://arxiv.org/abs/2606.00922", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Md Mainul Abrar", + "Xun Jia", + "Yujie Chi" + ], + "categories": [ + "physics.med-ph", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.00922", + "source": "arxiv", + "source_id": "arxiv:2606.00922", + "pdf_url": "https://arxiv.org/pdf/2606.00922", + "primary_query": "planning-agent" + }, + { + "id": "2606.07595", + "title": "VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents", + "url": "https://arxiv.org/abs/2606.07595", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Youting Wang", + "Yuan Tang", + "Yitian Qian", + "Chen Zhao" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.07595", + "source": "arxiv", + "source_id": "arxiv:2606.07595", + "pdf_url": "https://arxiv.org/pdf/2606.07595", + "primary_query": "language-agent" + }, + { + "id": "2606.00341", + "title": "ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use", + "url": "https://arxiv.org/abs/2606.00341", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Jeremy Tien", + "Abishek Anand", + "Yu-Rou Tuan", + "Yuchen Shen", + "J. Zico Kolter", + "Aran Nayebi" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.00341", + "source": "arxiv", + "source_id": "arxiv:2606.00341", + "pdf_url": "https://arxiv.org/pdf/2606.00341", + "primary_query": "agent-safety" + }, + { + "id": "2605.31377", + "title": "DynaTree: Dynamic Agentic Retrieval Tree for Time-Sensitive News Retrieval", + "url": "https://arxiv.org/abs/2605.31377", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Siyuan Qi", + "Xinyuan Wang", + "Yingxuan Yang", + "Haochuan Guo", + "Jianghao Lin", + "Weiwen Liu", + "Yong Yu", + "Weinan Zhang" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.31377", + "source": "arxiv", + "source_id": "arxiv:2605.31377", + "pdf_url": "https://arxiv.org/pdf/2605.31377", + "primary_query": "rag-agent" + }, + { + "id": "2605.30947", + "title": "Extending AI for Research to the Humanities: A Multi-Agent Framework for Evidence-Grounded Scholarship", + "url": "https://arxiv.org/abs/2605.30947", + "published": "2026-05-29", + "updated": "2026-06-03", + "authors": [ + "Yating Pan", + "Jiajun Zhang", + "Jun Wang", + "Qi Su" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.30947", + "source": "arxiv", + "source_id": "arxiv:2605.30947", + "pdf_url": "https://arxiv.org/pdf/2605.30947", + "primary_query": "rag-agent" + }, + { + "id": "2605.29960", + "title": "Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction", + "url": "https://arxiv.org/abs/2605.29960", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Hongtao Wang", + "Se Yang", + "Yu Chen", + "Puzhuo Liu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.29960", + "source": "arxiv", + "source_id": "arxiv:2605.29960", + "pdf_url": "https://arxiv.org/pdf/2605.29960", + "primary_query": "agent-memory" + }, + { + "id": "2605.25435", + "title": "Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures", + "url": "https://arxiv.org/abs/2605.25435", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Yuntao Wang", + "Jianle Ba", + "Han Liu", + "Yanghe Pan", + "Jintao Wei", + "Zhou Su", + "Tom H. Luan", + "Linkang Du" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.25435", + "source": "arxiv", + "source_id": "arxiv:2605.25435", + "pdf_url": "https://arxiv.org/pdf/2605.25435", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.24812", + "title": "CoRe-Code: Collaborative Reinforcement Learning for Code Generation", + "url": "https://arxiv.org/abs/2605.24812", + "published": "2026-05-24", + "updated": "2026-05-24", + "authors": [ + "Zhihao Dou", + "Qinjian Zhao", + "Zhongwei Wan", + "Xiaoyu Xia", + "Sumon Biswas" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.24812", + "source": "arxiv", + "source_id": "arxiv:2605.24812", + "pdf_url": "https://arxiv.org/pdf/2605.24812", + "primary_query": "planning-agent" + }, + { + "id": "2605.23067", + "title": "What Training Data Teaches RL Memory Agents: An Empirical Study of Curriculum Effects in Memory-Augmented QA", + "url": "https://arxiv.org/abs/2605.23067", + "published": "2026-05-21", + "updated": "2026-05-21", + "authors": [ + "Xinjie He", + "Zhiyuan Lin", + "Su Liu", + "Jialun Wu", + "Qiyang Xie", + "Weikai Zhou", + "Shuai Xiao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.23067", + "source": "arxiv", + "source_id": "arxiv:2605.23067", + "pdf_url": "https://arxiv.org/pdf/2605.23067", + "primary_query": "agent-memory" + }, + { + "id": "2605.20616", + "title": "Auto-Dreamer: Learning Offline Memory Consolidation for Language Agents", + "url": "https://arxiv.org/abs/2605.20616", + "published": "2026-05-20", + "updated": "2026-05-20", + "authors": [ + "Chongrui Ye", + "Yuxiang Liu", + "Yu Wang", + "Haofei Yu", + "Yining Zhao", + "Ge Liu", + "Julian McAuley", + "Jiaxuan You" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "language-agent" + ], + "arxiv_id": "2605.20616", + "source": "arxiv", + "source_id": "arxiv:2605.20616", + "pdf_url": "https://arxiv.org/pdf/2605.20616", + "primary_query": "agent-memory" + }, + { + "id": "2605.16233", + "title": "FORGE: Self-Evolving Agent Memory With No Weight Updates via Population Broadcast", + "url": "https://arxiv.org/abs/2605.16233", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Igor Bogdanov", + "Chung-Horng Lung", + "Thomas Kunz", + "Jie Gao", + "Adrian Taylor", + "Marzia Zaman" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA", + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.16233", + "source": "arxiv", + "source_id": "arxiv:2605.16233", + "pdf_url": "https://arxiv.org/pdf/2605.16233", + "primary_query": "agent-memory" + }, + { + "id": "2605.15759", + "title": "DimMem: Dimensional Structuring for Efficient Long-Term Agent Memory", + "url": "https://arxiv.org/abs/2605.15759", + "published": "2026-05-15", + "updated": "2026-05-24", + "authors": [ + "Wentao Qiu", + "Haotian Hu", + "Fanyi Wang", + "Jinwei Kong", + "Yu Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.15759", + "source": "arxiv", + "source_id": "arxiv:2605.15759", + "pdf_url": "https://arxiv.org/pdf/2605.15759", + "primary_query": "agent-memory" + }, + { + "id": "2605.15710", + "title": "SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory", + "url": "https://arxiv.org/abs/2605.15710", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Huacan Chai", + "Yukai Wang", + "Yingxuan Yang", + "Dan Peng", + "Yuanyi Song", + "Zhihui Fu", + "Weiwen Liu", + "Jianghao Lin", + "Jun Wang", + "Weinan Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.15710", + "source": "arxiv", + "source_id": "arxiv:2605.15710", + "pdf_url": "https://arxiv.org/pdf/2605.15710", + "primary_query": "agent-memory" + }, + { + "id": "2605.15128", + "title": "MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory", + "url": "https://arxiv.org/abs/2605.15128", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Minghao Guo", + "Qingyue Jiao", + "Zeru Shi", + "Yihao Quan", + "Boxuan Zhang", + "Danrui Li", + "Liwei Che", + "Wujiang Xu", + "Shilong Liu", + "Zirui Liu", + "Mubbasir Kapadia", + "Vladimir Pavlovic", + "Jiang Liu", + "Mengdi Wang", + "Yiyu Shi", + "Dimitris N. Metaxas", + "Ruixiang Tang" + ], + "categories": [ + "cs.CV", + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.15128", + "source": "arxiv", + "source_id": "arxiv:2605.15128", + "pdf_url": "https://arxiv.org/pdf/2605.15128", + "primary_query": "agent-memory" + }, + { + "id": "2605.14906", + "title": "MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models", + "url": "https://arxiv.org/abs/2605.14906", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Xiyu Ren", + "Zhaowei Wang", + "Yiming Du", + "Zhongwei Xie", + "Chi Liu", + "Xinlin Yang", + "Haoyue Feng", + "Wenjun Pan", + "Tianshi Zheng", + "Baixuan Xu", + "Zhengnan Li", + "Yangqiu Song", + "Ginny Wong", + "Simon See" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.14906", + "source": "arxiv", + "source_id": "arxiv:2605.14906", + "pdf_url": "https://arxiv.org/pdf/2605.14906", + "primary_query": "agent-memory" + }, + { + "id": "2605.14932", + "title": "Toward Securing AI Agents Like Operating Systems", + "url": "https://arxiv.org/abs/2605.14932", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Lukas Pirch", + "Micha Horlboge", + "Patrick Großmann", + "Syeda Mahnur Asif", + "Klim Kireev", + "Thorsten Holz", + "Konrad Rieck" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14932", + "source": "arxiv", + "source_id": "arxiv:2605.14932", + "pdf_url": "https://arxiv.org/pdf/2605.14932", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.11946", + "title": "Counterfactual Trace Auditing of LLM Agent Skills", + "url": "https://arxiv.org/abs/2605.11946", + "published": "2026-05-12", + "updated": "2026-05-28", + "authors": [ + "Xiaolin Zhou", + "Jinbo Liu", + "Li Li", + "Ryan A. Rossi", + "Xiyang Hu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.11946", + "source": "arxiv", + "source_id": "arxiv:2605.11946", + "pdf_url": "https://arxiv.org/pdf/2605.11946", + "primary_query": "planning-agent" + }, + { + "id": "2605.08964", + "title": "Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents", + "url": "https://arxiv.org/abs/2605.08964", + "published": "2026-05-09", + "updated": "2026-05-09", + "authors": [ + "Carol Xuan Long" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.08964", + "source": "arxiv", + "source_id": "arxiv:2605.08964", + "pdf_url": "https://arxiv.org/pdf/2605.08964", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.07830", + "title": "CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios", + "url": "https://arxiv.org/abs/2605.07830", + "published": "2026-05-08", + "updated": "2026-05-08", + "authors": [ + "Taein Lim", + "Seongyong Ju", + "Munhyeok Kim", + "Hyunjun Kim", + "Hoki Kim" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.07830", + "source": "arxiv", + "source_id": "arxiv:2605.07830", + "pdf_url": "https://arxiv.org/pdf/2605.07830", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.04808", + "title": "DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents", + "url": "https://arxiv.org/abs/2605.04808", + "published": "2026-05-06", + "updated": "2026-05-06", + "authors": [ + "Zhaorun Chen", + "Xun Liu", + "Haibo Tong", + "Chengquan Guo", + "Yuzhou Nie", + "Jiawei Zhang", + "Mintong Kang", + "Chejian Xu", + "Qichang Liu", + "Xiaogeng Liu", + "Tianneng Shi", + "Chaowei Xiao", + "Sanmi Koyejo", + "Percy Liang", + "Wenbo Guo", + "Dawn Song", + "Bo Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.04808", + "source": "arxiv", + "source_id": "arxiv:2605.04808", + "pdf_url": "https://arxiv.org/pdf/2605.04808", + "primary_query": "agent-safety" + }, + { + "id": "2605.01644", + "title": "Toward a Principled Framework for Agent Safety Measurement", + "url": "https://arxiv.org/abs/2605.01644", + "published": "2026-05-02", + "updated": "2026-05-02", + "authors": [ + "Shuyi Lin", + "Anshuman Suri", + "Alina Oprea", + "Cheng Tan" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.01644", + "source": "arxiv", + "source_id": "arxiv:2605.01644", + "pdf_url": "https://arxiv.org/pdf/2605.01644", + "primary_query": "agent-safety" + }, + { + "id": "2605.00741", + "title": "Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems", + "url": "https://arxiv.org/abs/2605.00741", + "published": "2026-05-01", + "updated": "2026-05-01", + "authors": [ + "Saeid Jamshidi", + "Foutse Khomh", + "Carol Fung", + "Kawser Wazed Nafi" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.00741", + "source": "arxiv", + "source_id": "arxiv:2605.00741", + "pdf_url": "https://arxiv.org/pdf/2605.00741", + "primary_query": "agent-safety" + }, + { + "id": "2604.24212", + "title": "Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis", + "url": "https://arxiv.org/abs/2604.24212", + "published": "2026-04-27", + "updated": "2026-04-27", + "authors": [ + "Jiahong Xiang", + "Xiaoyang Xu", + "Xiaopan Chu", + "Hongliang Tian", + "Yuqun Zhang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.24212", + "source": "arxiv", + "source_id": "arxiv:2604.24212", + "pdf_url": "https://arxiv.org/pdf/2604.24212", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.23210", + "title": "Discovering Agentic Safety Specifications from 1-Bit Danger Signals", + "url": "https://arxiv.org/abs/2604.23210", + "published": "2026-04-25", + "updated": "2026-04-25", + "authors": [ + "Víctor Gallego" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.23210", + "source": "arxiv", + "source_id": "arxiv:2604.23210", + "pdf_url": "https://arxiv.org/pdf/2604.23210", + "primary_query": "agent-safety" + }, + { + "id": "2604.13954", + "title": "HINTBench: Horizon-agent Intrinsic Non-attack Trajectory Benchmark", + "url": "https://arxiv.org/abs/2604.13954", + "published": "2026-04-15", + "updated": "2026-04-15", + "authors": [ + "Jiacheng Wang", + "Jinchang Hou", + "Fabian Wang", + "Ping Jian", + "Chenfu Bao", + "Zhonghou Lv" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.13954", + "source": "arxiv", + "source_id": "arxiv:2604.13954", + "pdf_url": "https://arxiv.org/pdf/2604.13954", + "primary_query": "agent-safety" + }, + { + "id": "2604.13536", + "title": "Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy", + "url": "https://arxiv.org/abs/2604.13536", + "published": "2026-04-15", + "updated": "2026-04-16", + "authors": [ + "Shawn Wanxiang Zhong", + "Junxuan Liao", + "Jing Liu", + "Mai Zheng", + "Andrea C. Arpaci-Dusseau", + "Remzi H. Arpaci-Dusseau" + ], + "categories": [ + "cs.OS" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.13536", + "source": "arxiv", + "source_id": "arxiv:2604.13536", + "pdf_url": "https://arxiv.org/pdf/2604.13536", + "primary_query": "agent-safety" + }, + { + "id": "2604.13298", + "title": "Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection", + "url": "https://arxiv.org/abs/2604.13298", + "published": "2026-04-14", + "updated": "2026-04-14", + "authors": [ + "Sujan Ghimire", + "Parsa Mirfasihi", + "Muhtasim Alam Chowdhury", + "Veeramani Pugazhenthi", + "Harish Kumar Dharavath", + "Farshad Firouzi", + "Rozhin Yasaei", + "Pratik Satam", + "Soheil Salehi" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.13298", + "source": "arxiv", + "source_id": "arxiv:2604.13298", + "pdf_url": "https://arxiv.org/pdf/2604.13298", + "primary_query": "agent-safety" + }, + { + "id": "2605.28835", + "title": "GenesisFunc: Multi-Agent Data Generation for Accurate and Generalizable Function-Calling", + "url": "https://arxiv.org/abs/2605.28835", + "published": "2026-04-10", + "updated": "2026-04-10", + "authors": [ + "Hao-Xiang Xu", + "Chong Deng", + "Jiaqing Liu", + "Wen Wang", + "Qian Chen", + "Lujia Bao", + "Xiangang Li", + "Zhen-Hua Ling" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.28835", + "source": "arxiv", + "source_id": "arxiv:2605.28835", + "pdf_url": "https://arxiv.org/pdf/2605.28835", + "primary_query": "function-calling" + }, + { + "id": "2605.00845", + "title": "Graph Query Generation with Constraint-guided Large Language Agents", + "url": "https://arxiv.org/abs/2605.00845", + "published": "2026-04-09", + "updated": "2026-04-09", + "authors": [ + "Mengying Wang", + "Nicolaas Jedema", + "Rahul Pandey", + "RaviKiran Krishnan", + "Jens Lehmann", + "Yinghui Wu" + ], + "categories": [ + "cs.DB", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.00845", + "source": "arxiv", + "source_id": "arxiv:2605.00845", + "pdf_url": "https://arxiv.org/pdf/2605.00845", + "primary_query": "language-agent" + }, + { + "id": "2604.26959", + "title": "CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs", + "url": "https://arxiv.org/abs/2604.26959", + "published": "2026-04-07", + "updated": "2026-04-07", + "authors": [ + "Elham Nasarian", + "Abhilash Neog", + "Kwok-Leung Tsui", + "Niyousha HosseiniChimeh" + ], + "categories": [ + "cs.CY", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.26959", + "source": "arxiv", + "source_id": "arxiv:2604.26959", + "pdf_url": "https://arxiv.org/pdf/2604.26959", + "primary_query": "agent-safety" + }, + { + "id": "2604.04131", + "title": "Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents", + "url": "https://arxiv.org/abs/2604.04131", + "published": "2026-04-05", + "updated": "2026-04-05", + "authors": [ + "Paulo Akira F. Enabe" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.04131", + "source": "arxiv", + "source_id": "arxiv:2604.04131", + "pdf_url": "https://arxiv.org/pdf/2604.04131", + "primary_query": "language-agent" + }, + { + "id": "2603.28428", + "title": "Synergy: A Next-Generation General-Purpose Agent for Open Agentic Web", + "url": "https://arxiv.org/abs/2603.28428", + "published": "2026-03-30", + "updated": "2026-03-30", + "authors": [ + "Xiaohang Nie", + "Zihan Guo", + "Kezhuo Yang", + "Zhichong Zheng", + "Bochen Ge", + "Shuai Pan", + "Zeyi Chen", + "Youling Xiang", + "Yu Zhang", + "Weiwen Liu", + "Yuanjian Zhou", + "Weinan Zhang" + ], + "categories": [ + "cs.CY", + "cs.MA" + ], + "topics": [ + "coding-agent", + "embodied-agent", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.28428", + "source": "arxiv", + "source_id": "arxiv:2603.28428", + "pdf_url": "https://arxiv.org/pdf/2603.28428", + "primary_query": "function-calling" + }, + { + "id": "2603.28166", + "title": "Evaluating Privilege Usage of Agents with Real-World Tools", + "url": "https://arxiv.org/abs/2603.28166", + "published": "2026-03-30", + "updated": "2026-04-20", + "authors": [ + "Quan Zhang", + "Lianhang Fu", + "Lvsi Lian", + "Gwihwan Go", + "Yujue Wang", + "Chijin Zhou", + "Yu Jiang", + "Geguang Pu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.28166", + "source": "arxiv", + "source_id": "arxiv:2603.28166", + "pdf_url": "https://arxiv.org/pdf/2603.28166", + "primary_query": "agent-safety" + }, + { + "id": "2603.21564", + "title": "Toward a Theory of Hierarchical Memory for Language Agents", + "url": "https://arxiv.org/abs/2603.21564", + "published": "2026-03-23", + "updated": "2026-03-23", + "authors": [ + "Yashar Talebirad", + "Ali Parsaee", + "Csongor Y. Szepesvari", + "Amirhossein Nadiri", + "Osmar Zaiane" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.IT", + "cs.SI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.21564", + "source": "arxiv", + "source_id": "arxiv:2603.21564", + "pdf_url": "https://arxiv.org/pdf/2603.21564", + "primary_query": "language-agent" + }, + { + "id": "2603.21357", + "title": "AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling", + "url": "https://arxiv.org/abs/2603.21357", + "published": "2026-03-22", + "updated": "2026-05-10", + "authors": [ + "Liang Ding" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "computer-use", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.21357", + "source": "arxiv", + "source_id": "arxiv:2603.21357", + "pdf_url": "https://arxiv.org/pdf/2603.21357", + "primary_query": "language-agent" + }, + { + "id": "2603.05553", + "title": "EigenData: A Self-Evolving Multi-Agent Platform for Function-Calling Data Synthesis, Auditing, and Repair", + "url": "https://arxiv.org/abs/2603.05553", + "published": "2026-03-05", + "updated": "2026-03-05", + "authors": [ + "Jiaao Chen", + "Jingyuan Qi", + "Mingye Gao", + "Wei-Chen Wang", + "Hanrui Wang", + "Di Jin" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.05553", + "source": "arxiv", + "source_id": "arxiv:2603.05553", + "pdf_url": "https://arxiv.org/pdf/2603.05553", + "primary_query": "function-calling" + }, + { + "id": "2602.07652", + "title": "Agent-Fence: Mapping Security Vulnerabilities Across Deep Research Agents", + "url": "https://arxiv.org/abs/2602.07652", + "published": "2026-02-07", + "updated": "2026-02-07", + "authors": [ + "Sai Puppala", + "Ismail Hossain", + "Md Jahangir Alam", + "Yoonpyo Lee", + "Jay Yoo", + "Tanzim Ahad", + "Syed Bahauddin Alam", + "Sajedul Talukder" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.07652", + "source": "arxiv", + "source_id": "arxiv:2602.07652", + "pdf_url": "https://arxiv.org/pdf/2602.07652", + "primary_query": "agent-safety" + }, + { + "id": "2602.05115", + "title": "SocialVeil: Probing Social Intelligence of Language Agents under Communication Barriers", + "url": "https://arxiv.org/abs/2602.05115", + "published": "2026-02-04", + "updated": "2026-02-04", + "authors": [ + "Keyang Xuan", + "Pengda Wang", + "Chongrui Ye", + "Haofei Yu", + "Tal August", + "Jiaxuan You" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.05115", + "source": "arxiv", + "source_id": "arxiv:2602.05115", + "pdf_url": "https://arxiv.org/pdf/2602.05115", + "primary_query": "language-agent" + }, + { + "id": "2602.03786", + "title": "AOrchestra: Automating Sub-Agent Creation for Agentic Orchestration", + "url": "https://arxiv.org/abs/2602.03786", + "published": "2026-02-03", + "updated": "2026-02-07", + "authors": [ + "Jianhao Ruan", + "Zhihao Xu", + "Yiran Peng", + "Fashen Ren", + "Zhaoyang Yu", + "Xinbing Liang", + "Jinyu Xiang", + "Yongru Chen", + "Bang Liu", + "Chenglin Wu", + "Yuyu Luo", + "Jiayi Zhang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.03786", + "source": "arxiv", + "source_id": "arxiv:2602.03786", + "pdf_url": "https://arxiv.org/pdf/2602.03786", + "primary_query": "language-agent" + }, + { + "id": "2601.00268", + "title": "Beyond Perfect APIs: A Comprehensive Evaluation of LLM Agents Under Real-World API Complexity", + "url": "https://arxiv.org/abs/2601.00268", + "published": "2026-01-01", + "updated": "2026-01-01", + "authors": [ + "Doyoung Kim", + "Zhiwei Ren", + "Jie Hao", + "Zhongkai Sun", + "Lichao Wang", + "Xiyao Ma", + "Zack Ye", + "Xu Han", + "Jun Yin", + "Heng Ji", + "Wei Shen", + "Xing Fan", + "Benjamin Yao", + "Chenlei Guo" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.00268", + "source": "arxiv", + "source_id": "arxiv:2601.00268", + "pdf_url": "https://arxiv.org/pdf/2601.00268", + "primary_query": "function-calling" + }, + { + "id": "2511.22138", + "title": "TinyLLM: Evaluation and Optimization of Small Language Models for Agentic Tasks on Edge Devices", + "url": "https://arxiv.org/abs/2511.22138", + "published": "2025-11-27", + "updated": "2025-11-27", + "authors": [ + "Mohd Ariful Haque", + "Fahad Rahman", + "Kishor Datta Gupta", + "Khalil Shujaee", + "Roy George" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2511.22138", + "source": "arxiv", + "source_id": "arxiv:2511.22138", + "pdf_url": "https://arxiv.org/pdf/2511.22138", + "primary_query": "function-calling" + }, + { + "id": "2511.11169", + "title": "Refine and Align: Confidence Calibration through Multi-Agent Interaction in VQA", + "url": "https://arxiv.org/abs/2511.11169", + "published": "2025-11-14", + "updated": "2025-11-14", + "authors": [ + "Ayush Pandey", + "Jai Bardhan", + "Ishita Jain", + "Ramya S Hebbalaguppe", + "Rohan Raju Dhanakshirur", + "Lovekesh Vig" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2511.11169", + "source": "arxiv", + "source_id": "arxiv:2511.11169", + "pdf_url": "https://arxiv.org/pdf/2511.11169", + "primary_query": "function-calling" + }, + { + "id": "2510.21524", + "title": "EU-Agent-Bench: Measuring Illegal Behavior of LLM Agents Under EU Law", + "url": "https://arxiv.org/abs/2510.21524", + "published": "2025-10-24", + "updated": "2025-10-24", + "authors": [ + "Ilija Lichkovski", + "Alexander Müller", + "Mariam Ibrahim", + "Tiwai Mhundwa" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.21524", + "source": "arxiv", + "source_id": "arxiv:2510.21524", + "pdf_url": "https://arxiv.org/pdf/2510.21524", + "primary_query": "function-calling" + }, + { + "id": "2509.08863", + "title": "GeoJSON Agents:A Multi-Agent LLM Architecture for Geospatial Analysis-Function Calling vs Code Generation", + "url": "https://arxiv.org/abs/2509.08863", + "published": "2025-09-10", + "updated": "2025-12-03", + "authors": [ + "Qianqian Luo", + "Qingming Lin", + "Liuchang Xu", + "Sensen Wu", + "Ruichen Mao", + "Chao Wang", + "Hailin Feng", + "Bo Huang", + "Zhenhong Du" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.08863", + "source": "arxiv", + "source_id": "arxiv:2509.08863", + "pdf_url": "https://arxiv.org/pdf/2509.08863", + "primary_query": "function-calling" + }, + { + "id": "2508.11027", + "title": "Hell or High Water: Evaluating Agentic Recovery from External Failures", + "url": "https://arxiv.org/abs/2508.11027", + "published": "2025-08-14", + "updated": "2025-08-14", + "authors": [ + "Andrew Wang", + "Sophia Hager", + "Adi Asija", + "Daniel Khashabi", + "Nicholas Andrews" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.11027", + "source": "arxiv", + "source_id": "arxiv:2508.11027", + "pdf_url": "https://arxiv.org/pdf/2508.11027", + "primary_query": "function-calling" + }, + { + "id": "2607.06223", + "title": "Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents", + "url": "https://arxiv.org/abs/2607.06223", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Yijun Zhang", + "Fan Xu", + "Jiaxin Ding", + "Yule Xie", + "Shiqing Gao", + "Xin Ding", + "Haoxiang Zhang", + "Luoyi Fu", + "Xinbing Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.06223", + "source": "arxiv", + "source_id": "arxiv:2607.06223", + "pdf_url": "https://arxiv.org/pdf/2607.06223", + "primary_query": "llm-agent" + }, + { + "id": "2607.05915", + "title": "PCBWorld: A Benchmark Environment for Engine-Grounded PCB Design Automation", + "url": "https://arxiv.org/abs/2607.05915", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Hyungseok Song", + "Junseok Park", + "Won-Seok Choi", + "Seohui Bae", + "Han-Seul Jeong", + "Youngjoon Park", + "Soonyoung Lee" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.05915", + "source": "arxiv", + "source_id": "arxiv:2607.05915", + "pdf_url": "https://arxiv.org/pdf/2607.05915", + "primary_query": "agentic-ai" + }, + { + "id": "2607.05805", + "title": "Onnes: A Physics-Grounded Multi-Agent LLM Simulator for Cryogenic Fault Diagnosis in Quantum Computing Infrastructure", + "url": "https://arxiv.org/abs/2607.05805", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Praneeth Narisetty", + "Uday Kumar Reddy Kattamanchi", + "Shiva Nagendra Babu Kore" + ], + "categories": [ + "cs.AI", + "cs.LG", + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.05805", + "source": "arxiv", + "source_id": "arxiv:2607.05805", + "pdf_url": "https://arxiv.org/pdf/2607.05805", + "primary_query": "llm-agent" + }, + { + "id": "2607.05743", + "title": "The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities", + "url": "https://arxiv.org/abs/2607.05743", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Mohammadreza Rashidi" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2607.05743", + "source": "arxiv", + "source_id": "arxiv:2607.05743", + "pdf_url": "https://arxiv.org/pdf/2607.05743", + "primary_query": "agentic-ai" + }, + { + "id": "2607.05690", + "title": "Memory in the Loop: In-Process Retrieval as ExtendedWorking Memory for Language Agents", + "url": "https://arxiv.org/abs/2607.05690", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Yusuf Khan", + "Carlo Lipizzi" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2607.05690", + "source": "arxiv", + "source_id": "arxiv:2607.05690", + "pdf_url": "https://arxiv.org/pdf/2607.05690", + "primary_query": "language-agent" + }, + { + "id": "2607.05055", + "title": "Toward Trustworthy Large Language Model Agents in Healthcare", + "url": "https://arxiv.org/abs/2607.05055", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Hadi Hasan", + "Safaa Salman", + "Adam Tai Abou Dargham", + "Ammar Mohanna", + "Ali Chehab" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling", + "rag-agent" + ], + "arxiv_id": "2607.05055", + "source": "arxiv", + "source_id": "arxiv:2607.05055", + "pdf_url": "https://arxiv.org/pdf/2607.05055", + "primary_query": "function-calling" + }, + { + "id": "2607.04089", + "title": "PLACEMEM: Toward a Compute-Aware Memory Plane for Lifelong Agents", + "url": "https://arxiv.org/abs/2607.04089", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Sukanta Ganguly" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2607.04089", + "source": "arxiv", + "source_id": "arxiv:2607.04089", + "pdf_url": "https://arxiv.org/pdf/2607.04089", + "primary_query": "agent-memory" + }, + { + "id": "2607.03702", + "title": "Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry", + "url": "https://arxiv.org/abs/2607.03702", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Weiyang Guo", + "Zesheng Shi", + "Longhui Zhang", + "Zeen Zhu", + "Min Zhang", + "Jing Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.03702", + "source": "arxiv", + "source_id": "arxiv:2607.03702", + "pdf_url": "https://arxiv.org/pdf/2607.03702", + "primary_query": "llm-agent" + }, + { + "id": "2607.03695", + "title": "Social Networks of LLM Agents", + "url": "https://arxiv.org/abs/2607.03695", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Kaixuan Liu", + "Guojun Xiong", + "Weinan Zhang", + "Shengpu Tang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.03695", + "source": "arxiv", + "source_id": "arxiv:2607.03695", + "pdf_url": "https://arxiv.org/pdf/2607.03695", + "primary_query": "llm-agent" + }, + { + "id": "2607.03821", + "title": "DualView: Preventing Indirect Prompt Injection in Personal AI Agents", + "url": "https://arxiv.org/abs/2607.03821", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Juhee Kim", + "Woohyuk Choi", + "Taehyun Kang", + "Youngmin Kim", + "Byoungyoung Lee" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.03821", + "source": "arxiv", + "source_id": "arxiv:2607.03821", + "pdf_url": "https://arxiv.org/pdf/2607.03821", + "primary_query": "ai-agent" + }, + { + "id": "2607.04009", + "title": "PhysMiner: An Agentic AI Framework for Discovering Turbulence Physics", + "url": "https://arxiv.org/abs/2607.04009", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Jiawei Chen", + "Han Gao", + "Ping He" + ], + "categories": [ + "physics.flu-dyn" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04009", + "source": "arxiv", + "source_id": "arxiv:2607.04009", + "pdf_url": "https://arxiv.org/pdf/2607.04009", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03968", + "title": "Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents", + "url": "https://arxiv.org/abs/2607.03968", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Abhishek Kumar", + "Carsten Maple" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03968", + "source": "arxiv", + "source_id": "arxiv:2607.03968", + "pdf_url": "https://arxiv.org/pdf/2607.03968", + "primary_query": "coding-agent" + }, + { + "id": "2607.02846", + "title": "Object-Centric Environment Modeling for Agentic Tasks", + "url": "https://arxiv.org/abs/2607.02846", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Yiyang Li", + "Tianyi Ma", + "Zehong Wang", + "Yijun Ma", + "Yanfang Ye" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.02846", + "source": "arxiv", + "source_id": "arxiv:2607.02846", + "pdf_url": "https://arxiv.org/pdf/2607.02846", + "primary_query": "llm-agent" + }, + { + "id": "2607.03269", + "title": "Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks", + "url": "https://arxiv.org/abs/2607.03269", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Haoxiang Luo", + "Yinqiu Liu", + "Ruichen Zhang", + "Guangyuan Liu", + "Gang Sun", + "Hongfang Yu", + "Zhu Han", + "Dong In Kim" + ], + "categories": [ + "cs.NI" + ], + "topics": [ + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.03269", + "source": "arxiv", + "source_id": "arxiv:2607.03269", + "pdf_url": "https://arxiv.org/pdf/2607.03269", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02911", + "title": "CoACT: Action-Preserving Observation Compression for Coding Agents", + "url": "https://arxiv.org/abs/2607.02911", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Haorui Chen", + "Yuancheng Zhu", + "Yitong Zhang", + "Jia Li" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02911", + "source": "arxiv", + "source_id": "arxiv:2607.02911", + "pdf_url": "https://arxiv.org/pdf/2607.02911", + "primary_query": "coding-agent" + }, + { + "id": "2607.01767", + "title": "Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts", + "url": "https://arxiv.org/abs/2607.01767", + "published": "2026-07-02", + "updated": "2026-07-05", + "authors": [ + "Xinyuan Song", + "Zekun Cai" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2607.01767", + "source": "arxiv", + "source_id": "arxiv:2607.01767", + "pdf_url": "https://arxiv.org/pdf/2607.01767", + "primary_query": "language-agent" + }, + { + "id": "2607.01709", + "title": "COMFYCLAW: Self-Evolving Skill Harnesses for Image Generation Workflows", + "url": "https://arxiv.org/abs/2607.01709", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Zongxia Li", + "Dawei Liu", + "Fuxiao Liu", + "Yuhang Zhou", + "Xiyang Wu", + "Jingxi Chen", + "Jing Xie", + "Xiaomin Wu", + "Lichao Sun" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2607.01709", + "source": "arxiv", + "source_id": "arxiv:2607.01709", + "pdf_url": "https://arxiv.org/pdf/2607.01709", + "primary_query": "agent-memory" + }, + { + "id": "2607.02807", + "title": "SwarmResearch: Orchestrating Coding Agents for Open-Ended Discovery", + "url": "https://arxiv.org/abs/2607.02807", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Yuvraj Virk", + "Zack Edds", + "Chunqiu Steven Xia", + "Lingming Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "coding-agent", + "computer-use", + "multi-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.02807", + "source": "arxiv", + "source_id": "arxiv:2607.02807", + "pdf_url": "https://arxiv.org/pdf/2607.02807", + "primary_query": "coding-agent" + }, + { + "id": "2607.02448", + "title": "AgentsCAD: Automated Design for Manufacturing of FDM Parts via Multi-Agent LLM Reasoning and Geometric Feature Recognition", + "url": "https://arxiv.org/abs/2607.02448", + "published": "2026-07-02", + "updated": "2026-07-07", + "authors": [ + "Emmanuel George", + "Christopher Keefe", + "Peter Pak", + "Amir Barati Farimani" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.02448", + "source": "arxiv", + "source_id": "arxiv:2607.02448", + "pdf_url": "https://arxiv.org/pdf/2607.02448", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.01213", + "title": "RepoRescue: An Empirical Study of LLM Agents on Whole-Repository Compatibility Rescue", + "url": "https://arxiv.org/abs/2607.01213", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Zhihao Lin", + "Mingyi Zhou", + "Zhensu Sun", + "Yizhuo Yang", + "Renyu Yang", + "David Lo", + "Li Li" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01213", + "source": "arxiv", + "source_id": "arxiv:2607.01213", + "pdf_url": "https://arxiv.org/pdf/2607.01213", + "primary_query": "llm-agent" + }, + { + "id": "2607.01120", + "title": "Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents", + "url": "https://arxiv.org/abs/2607.01120", + "published": "2026-07-01", + "updated": "2026-07-02", + "authors": [ + "Ran Yan", + "Wei Fu", + "Jiale Li", + "Shusheng Xu", + "Zhiyu Mei", + "Jiaxuan Gao", + "Jiarui Zhang", + "Wentai Zhang", + "Hao Dai", + "Xujie Shen", + "Chuyi He", + "Zhen Pu", + "Jun Mei", + "Zhiyao Lin", + "Haitao Wang", + "Zhiqiang Ding", + "Jiawei Zhang", + "Huaijie Wang", + "Ruida Xu", + "Honghua Dong", + "Youhe Jiang", + "Yi Wu", + "Tongkai Yang", + "Binhang Yuan" + ], + "categories": [ + "cs.DC" + ], + "topics": [ + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01120", + "source": "arxiv", + "source_id": "arxiv:2607.01120", + "pdf_url": "https://arxiv.org/pdf/2607.01120", + "primary_query": "llm-agent" + }, + { + "id": "2607.00345", + "title": "Registry-Governed Agent Lifecycle:Completing EDDOps with Evaluation-DrivenRegistration, Promotion, and Retirement on AWS AgentCore", + "url": "https://arxiv.org/abs/2607.00345", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Richard Kang", + "Vincent Wang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00345", + "source": "arxiv", + "source_id": "arxiv:2607.00345", + "pdf_url": "https://arxiv.org/pdf/2607.00345", + "primary_query": "llm-agent" + }, + { + "id": "2607.00911", + "title": "From Registry to Repository: How AI Agent Skills Are Written, Adapted, and Maintained", + "url": "https://arxiv.org/abs/2607.00911", + "published": "2026-07-01", + "updated": "2026-07-06", + "authors": [ + "Haoyu Gao", + "Jai Lal Lulla", + "Hong Yi Lin", + "Sebastian Baltes", + "Christoph Treude", + "Mansooreh Zahedi" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "coding-agent", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2607.00911", + "source": "arxiv", + "source_id": "arxiv:2607.00911", + "pdf_url": "https://arxiv.org/pdf/2607.00911", + "primary_query": "ai-agent" + }, + { + "id": "2607.01421", + "title": "Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance", + "url": "https://arxiv.org/abs/2607.01421", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Laxmipriya Ganesh Iyer" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.01421", + "source": "arxiv", + "source_id": "arxiv:2607.01421", + "pdf_url": "https://arxiv.org/pdf/2607.01421", + "primary_query": "agentic-ai" + }, + { + "id": "2607.01061", + "title": "Agentic generation of verifiable rules for deterministic, self-expanding reaction classification", + "url": "https://arxiv.org/abs/2607.01061", + "published": "2026-07-01", + "updated": "2026-07-05", + "authors": [ + "Daniel Armstrong", + "Maarten Dobbelaere", + "Valentas Olikauskas", + "Helena Avila", + "Octavian Susanu", + "Jérôme Waser", + "Philippe Schwaller" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "coding-agent", + "multi-agent", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.01061", + "source": "arxiv", + "source_id": "arxiv:2607.01061", + "pdf_url": "https://arxiv.org/pdf/2607.01061", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31744", + "title": "A Conversational Agentic Interface to Physics-Based Household Digital Twins for Residential Energy Decision Support", + "url": "https://arxiv.org/abs/2606.31744", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Costas Mylonas", + "Titos Georgoulakis", + "Magda Foti" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.31744", + "source": "arxiv", + "source_id": "arxiv:2606.31744", + "pdf_url": "https://arxiv.org/pdf/2606.31744", + "primary_query": "llm-agent" + }, + { + "id": "2606.31209", + "title": "Long-term Traffic Simulation via Structured Autoregressive Modeling", + "url": "https://arxiv.org/abs/2606.31209", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Lingyu Xiao", + "Zexin Feng", + "Xintao Yan" + ], + "categories": [ + "cs.AI", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31209", + "source": "arxiv", + "source_id": "arxiv:2606.31209", + "pdf_url": "https://arxiv.org/pdf/2606.31209", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31085", + "title": "DDIAgents: Mechanism-Conditioned Context Flow for Drug-Drug Interaction Prediction", + "url": "https://arxiv.org/abs/2606.31085", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Zhenqian Shen", + "Yu Liu", + "Xiaoyi Fu", + "Quanming Yao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31085", + "source": "arxiv", + "source_id": "arxiv:2606.31085", + "pdf_url": "https://arxiv.org/pdf/2606.31085", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.30383", + "title": "Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents", + "url": "https://arxiv.org/abs/2606.30383", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Bojie Li", + "Noah Shi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.30383", + "source": "arxiv", + "source_id": "arxiv:2606.30383", + "pdf_url": "https://arxiv.org/pdf/2606.30383", + "primary_query": "llm-agent" + }, + { + "id": "2606.30697", + "title": "LUMOS: A Semantic Operating-System Layer for Accessibility-Grounded AI Agents", + "url": "https://arxiv.org/abs/2606.30697", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yogeswar Reddy Thota" + ], + "categories": [ + "cs.OS", + "cs.AI", + "cs.CV" + ], + "topics": [ + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "web-gui-agent" + ], + "arxiv_id": "2606.30697", + "source": "arxiv", + "source_id": "arxiv:2606.30697", + "pdf_url": "https://arxiv.org/pdf/2606.30697", + "primary_query": "ai-agent" + }, + { + "id": "2606.30877", + "title": "A Systematic Approach to Multi-Agent AI from Advanced Regulatory Control Theory: Safe and Auditable LLM Operator Agents for Process Control", + "url": "https://arxiv.org/abs/2606.30877", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Idelfonso B. R. Nogueira", + "Sigurd Skogestad" + ], + "categories": [ + "eess.SY", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.30877", + "source": "arxiv", + "source_id": "arxiv:2606.30877", + "pdf_url": "https://arxiv.org/pdf/2606.30877", + "primary_query": "agentic-ai" + }, + { + "id": "2606.29957", + "title": "SWE-Together: Evaluating Coding Agents in Interactive User Sessions", + "url": "https://arxiv.org/abs/2606.29957", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yifan Wu", + "Zhuokai Zhao", + "Songlin Li", + "Ho Hin Lee", + "Jiacheng Zhu", + "Shirley Wu", + "Tianhe Yu", + "Serena Li", + "Lizhu Zhang", + "Xiangjun Fan", + "Shengzhi Li" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2606.29957", + "source": "arxiv", + "source_id": "arxiv:2606.29957", + "pdf_url": "https://arxiv.org/pdf/2606.29957", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.29778", + "title": "Mandol: An Agglomerative Agent Memory System for Long-Term Conversations", + "url": "https://arxiv.org/abs/2606.29778", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yuhan Zhang", + "Zhiyuan Guo", + "Ziheng Zeng", + "Wei Wang", + "Wentao Wu", + "Lijie Xu" + ], + "categories": [ + "cs.DB", + "cs.AI", + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2606.29778", + "source": "arxiv", + "source_id": "arxiv:2606.29778", + "pdf_url": "https://arxiv.org/pdf/2606.29778", + "primary_query": "agent-memory" + }, + { + "id": "2606.30573", + "title": "SWE-INTERACT: Reimagining SWE Benchmarks as User-Driven Long-Horizon Coding Sessions", + "url": "https://arxiv.org/abs/2606.30573", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Mohit Raghavendra", + "Anisha Gunjal", + "Aakash Sabharwal", + "Yunzhong He" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.30573", + "source": "arxiv", + "source_id": "arxiv:2606.30573", + "pdf_url": "https://arxiv.org/pdf/2606.30573", + "primary_query": "coding-agent" + }, + { + "id": "2606.30560", + "title": "TraceLab: Characterizing Coding Agent Workloads for LLM Serving", + "url": "https://arxiv.org/abs/2606.30560", + "published": "2026-06-29", + "updated": "2026-06-30", + "authors": [ + "Kan Zhu", + "Mathew Jacob", + "Chenxi Ma", + "Yi Pan", + "Stephanie Wang", + "Arvind Krishnamurthy", + "Baris Kasikci" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.PF" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.30560", + "source": "arxiv", + "source_id": "arxiv:2606.30560", + "pdf_url": "https://arxiv.org/pdf/2606.30560", + "primary_query": "coding-agent" + }, + { + "id": "2606.30887", + "title": "Training Therapeutic Judges and Multi-Agent Systems for Human-Aligned Mental Health Support", + "url": "https://arxiv.org/abs/2606.30887", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Mizanur Rahman", + "Abeer Badawi", + "Elahe Rahimi", + "Laleh Seyyed-Kalantari", + "Frank Rudzicz", + "Enamul Hoque", + "Elham Dolatabadi" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30887", + "source": "arxiv", + "source_id": "arxiv:2606.30887", + "pdf_url": "https://arxiv.org/pdf/2606.30887", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00038", + "title": "Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting", + "url": "https://arxiv.org/abs/2607.00038", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Sandeco Macedo" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.00038", + "source": "arxiv", + "source_id": "arxiv:2607.00038", + "pdf_url": "https://arxiv.org/pdf/2607.00038", + "primary_query": "coding-agent" + }, + { + "id": "2606.29014", + "title": "Customized Generative AI Agent for Transportation Engineering Practice: A Development and Continued Pre-training Guideline", + "url": "https://arxiv.org/abs/2606.29014", + "published": "2026-06-27", + "updated": "2026-07-03", + "authors": [ + "Dianwei Chen", + "Yuan-Zheng Lei", + "Zifan Zhang", + "Yuchen Liu", + "Xianfeng Yang" + ], + "categories": [ + "cs.AI", + "cs.DL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.29014", + "source": "arxiv", + "source_id": "arxiv:2606.29014", + "pdf_url": "https://arxiv.org/pdf/2606.29014", + "primary_query": "ai-agent" + }, + { + "id": "2606.28781", + "title": "HyphaeDB: A Living Knowledge Topology for Agent-First Memory", + "url": "https://arxiv.org/abs/2606.28781", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Krishna Halaharvi" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "coding-agent", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "agentic-ai" + ], + "arxiv_id": "2606.28781", + "source": "arxiv", + "source_id": "arxiv:2606.28781", + "pdf_url": "https://arxiv.org/pdf/2606.28781", + "primary_query": "agent-memory" + }, + { + "id": "2606.28839", + "title": "The Contagion Tensor: A Framework for Measuring Output-Distribution Coupling in Multi-Agent LLM Systems -- and Auditing the Claims It Enables", + "url": "https://arxiv.org/abs/2606.28839", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Zewen Liu" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28839", + "source": "arxiv", + "source_id": "arxiv:2606.28839", + "pdf_url": "https://arxiv.org/pdf/2606.28839", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28270", + "title": "Agent-Native Immune System: Architecture, Taxonomy, and Engineering", + "url": "https://arxiv.org/abs/2606.28270", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Bo Shen", + "Lifeng Chang", + "Tianyuan Wei", + "Yunpeng Li", + "Feng Shi", + "Yichen Han", + "Peijie Gao", + "Shiyi Kuang", + "Xin Chang", + "Dehui Li" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.28270", + "source": "arxiv", + "source_id": "arxiv:2606.28270", + "pdf_url": "https://arxiv.org/pdf/2606.28270", + "primary_query": "tool-use" + }, + { + "id": "2606.28436", + "title": "Dockerless: Environment-Free Program Verifier for Coding Agents", + "url": "https://arxiv.org/abs/2606.28436", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Wenhao Zeng", + "Yuling Shi", + "Xiaodong Gu", + "Chao Hu", + "Chaofan Wang", + "Yuhao Cui", + "Hongting Zhou", + "Mengnan Qi", + "Jianqiao Wangni", + "Zhaojian Yu", + "Shuzheng Gao", + "Kai Cai", + "Shilin He" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.28436", + "source": "arxiv", + "source_id": "arxiv:2606.28436", + "pdf_url": "https://arxiv.org/pdf/2606.28436", + "primary_query": "coding-agent" + }, + { + "id": "2606.26918", + "title": "Diagnosing Task Insensitivity in Language Agents", + "url": "https://arxiv.org/abs/2606.26918", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Jingyu Liu", + "Xiaopeng Wu", + "Kehan Chen", + "Chuan Yu", + "Yong Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.26918", + "source": "arxiv", + "source_id": "arxiv:2606.26918", + "pdf_url": "https://arxiv.org/pdf/2606.26918", + "primary_query": "language-agent" + }, + { + "id": "2606.26524", + "title": "VIGIL: Runtime Enforcement of Behavioral Specifications in AI Agent Skills", + "url": "https://arxiv.org/abs/2606.26524", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Ying Li", + "Yanju Chen", + "Hongbo Wen", + "Bosi Zhang", + "Hanzhi Liu", + "Peiran Wang", + "Yu Feng", + "Yuan Tian" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.26524", + "source": "arxiv", + "source_id": "arxiv:2606.26524", + "pdf_url": "https://arxiv.org/pdf/2606.26524", + "primary_query": "ai-agent" + }, + { + "id": "2606.27499", + "title": "DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection", + "url": "https://arxiv.org/abs/2606.27499", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Yujin Tang", + "Chenming Shang", + "Ruize Xu", + "Nikhil Singh" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.27499", + "source": "arxiv", + "source_id": "arxiv:2606.27499", + "pdf_url": "https://arxiv.org/pdf/2606.27499", + "primary_query": "agent-memory" + }, + { + "id": "2606.27243", + "title": "NOVA: A Verification-Aware Agent Harness for Architecture Evolution in Industrial Recommender Systems", + "url": "https://arxiv.org/abs/2606.27243", + "published": "2026-06-25", + "updated": "2026-06-26", + "authors": [ + "Shaohua Liu", + "Liang Fang", + "Yilong Sun", + "Shudong Huang", + "Qingsong Luo", + "Shaoxin Liu", + "Xiaoyang Chen", + "Dongqiang Liu", + "Chuangang Ma", + "Zhenzhen Chai", + "Henghuan Wang", + "Shijie Quan", + "Changyuan Cui", + "Zhangbin Zhu", + "Peng Chen", + "Wei Xu", + "Lei Xiao", + "Haijie Gu", + "Jie Jiang" + ], + "categories": [ + "cs.IR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "memory", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.27243", + "source": "arxiv", + "source_id": "arxiv:2606.27243", + "pdf_url": "https://arxiv.org/pdf/2606.27243", + "primary_query": "coding-agent" + }, + { + "id": "2606.26924", + "title": "A Deterministic Control Plane for LLM Coding Agents", + "url": "https://arxiv.org/abs/2606.26924", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Padmaraj Madatha" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.26924", + "source": "arxiv", + "source_id": "arxiv:2606.26924", + "pdf_url": "https://arxiv.org/pdf/2606.26924", + "primary_query": "coding-agent" + }, + { + "id": "2606.26883", + "title": "EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents", + "url": "https://arxiv.org/abs/2606.26883", + "published": "2026-06-25", + "updated": "2026-06-30", + "authors": [ + "Ryuji Hashimoto", + "Masahiro Kaneko", + "Kentaro Ueda", + "Takehiro Takayanagi", + "Kiyoshi Izumi" + ], + "categories": [ + "cs.DL" + ], + "topics": [ + "memory", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.26883", + "source": "arxiv", + "source_id": "arxiv:2606.26883", + "pdf_url": "https://arxiv.org/pdf/2606.26883", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28409", + "title": "Evidence-Driven LLM Agent for C-to-Synthesizable-C Conversion and Verification", + "url": "https://arxiv.org/abs/2606.28409", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Zhe Zhao", + "Hongbing Lang", + "Zhihan Xiao", + "Luke Ztz Hu", + "John Imoleayo Adebisi", + "Songping Mai" + ], + "categories": [ + "cs.AR", + "cs.AI" + ], + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.28409", + "source": "arxiv", + "source_id": "arxiv:2606.28409", + "pdf_url": "https://arxiv.org/pdf/2606.28409", + "primary_query": "rag-agent" + }, + { + "id": "2606.25819", + "title": "Beyond Function Calling: Benchmarking Tool-Using Agents under Tool-Environment Unreliability", + "url": "https://arxiv.org/abs/2606.25819", + "published": "2026-06-24", + "updated": "2026-06-27", + "authors": [ + "Yang Tian", + "Zhengpeng Shi", + "Yu Zhou", + "Bo Zhao" + ], + "categories": [ + "cs.CL", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling", + "tool-use" + ], + "arxiv_id": "2606.25819", + "source": "arxiv", + "source_id": "arxiv:2606.25819", + "pdf_url": "https://arxiv.org/pdf/2606.25819", + "primary_query": "function-calling" + }, + { + "id": "2606.26453", + "title": "Optimizing CUDA like a Human: Micro-Profiling Tools as Expert Surrogates for LLM-Based GPU Kernel Optimization", + "url": "https://arxiv.org/abs/2606.26453", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Jiading Gai", + "Shuai Zhang", + "Kaj Bostrom", + "Jin Huang", + "Vihang Patil", + "Haoyang Fang", + "Bernie Wang", + "Huzefa Rangwala", + "George Karypis" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.26453", + "source": "arxiv", + "source_id": "arxiv:2606.26453", + "pdf_url": "https://arxiv.org/pdf/2606.26453", + "primary_query": "coding-agent" + }, + { + "id": "2606.25361", + "title": "Memory Makes the Difference: Evaluating How Different Memory Roles Shape Conversational Agents", + "url": "https://arxiv.org/abs/2606.25361", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yuxin Wang", + "Paul Thomas", + "Zhiwei Yu", + "Yuan Gao", + "Saeed Hassanpour", + "Soroush Vosoughi", + "Robert Sim", + "Nick Craswell" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.25361", + "source": "arxiv", + "source_id": "arxiv:2606.25361", + "pdf_url": "https://arxiv.org/pdf/2606.25361", + "primary_query": "rag-agent" + }, + { + "id": "2606.27397", + "title": "SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game", + "url": "https://arxiv.org/abs/2606.27397", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yeqi Feng", + "Yuxin Chen", + "Tianxing He" + ], + "categories": [ + "cs.MA", + "cs.AI", + "cs.GT" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.27397", + "source": "arxiv", + "source_id": "arxiv:2606.27397", + "pdf_url": "https://arxiv.org/pdf/2606.27397", + "primary_query": "planning-agent" + }, + { + "id": "2606.25139", + "title": "Buildrix: An Open Platform for Sharing and Benchmarking Agentic AI Skills in Building Engineering", + "url": "https://arxiv.org/abs/2606.25139", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Zixin Jiang", + "Bing Dong" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.25139", + "source": "arxiv", + "source_id": "arxiv:2606.25139", + "pdf_url": "https://arxiv.org/pdf/2606.25139", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24779", + "title": "DeepBD: A Grounded Agentic Workflow for Variant Prioritization and Diagnosis of Genetic Birth Defects", + "url": "https://arxiv.org/abs/2606.24779", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Shiyu Li", + "Ziqi Yan", + "Zhihao Wu", + "Jielong Lu", + "Weiran Liao", + "Jiajun Yu", + "Genjie Li", + "Zeyu Chu", + "Jiajun Bu", + "Haishuai Wang" + ], + "categories": [ + "q-bio.GN", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.24779", + "source": "arxiv", + "source_id": "arxiv:2606.24779", + "pdf_url": "https://arxiv.org/pdf/2606.24779", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24193", + "title": "SkyChain Intelligence: A Blockchain-Secured Multi-Agent DRL Framework for Low-Altitude Embodied Artificial Intelligence", + "url": "https://arxiv.org/abs/2606.24193", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Haoxiang Luo", + "Tianqi Jiang", + "Ruichen Zhang", + "Yinqiu Liu", + "Gang Sun", + "Hongfang Yu", + "Abbas Jamalipour", + "Dong In Kim" + ], + "categories": [ + "cs.NI", + "cs.DC" + ], + "topics": [ + "agent-safety", + "embodied-agent", + "multi-agent", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.24193", + "source": "arxiv", + "source_id": "arxiv:2606.24193", + "pdf_url": "https://arxiv.org/pdf/2606.24193", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24597", + "title": "Qwen-AgentWorld: Language World Models for General Agents", + "url": "https://arxiv.org/abs/2606.24597", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yuxin Zuo", + "Zikai Xiao", + "Li Sheng", + "Fei Huang", + "Jianhong Tu", + "Yuxuan Liu", + "Tianyi Tang", + "Xiaomeng Hu", + "Yang Su", + "Qingfeng Lan", + "Yantao Liu", + "Qin Zhu", + "Yinger Zhang", + "Bowen Yu", + "Haiquan Zhao", + "Haiyang Xu", + "Jianxin Yang", + "Jiayang Cheng", + "Junyang Wang", + "Lianghao Deng", + "Mingfeng Xue", + "Tianyi Bai", + "Yang Fan", + "Yubo Ma", + "Yucheng Li", + "Zeyu Cui", + "Zhihai Wang", + "Zhihui Xie", + "Zhuorui Ye", + "An Yang", + "Dayiheng Liu", + "Jingren Zhou", + "Ning Ding" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.24597", + "source": "arxiv", + "source_id": "arxiv:2606.24597", + "pdf_url": "https://arxiv.org/pdf/2606.24597", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.25115", + "title": "Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory", + "url": "https://arxiv.org/abs/2606.25115", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Beining Wu", + "Zihao Ding", + "Jun Huang", + "Yanxiao Zhao" + ], + "categories": [ + "cs.LG", + "cs.NI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.25115", + "source": "arxiv", + "source_id": "arxiv:2606.25115", + "pdf_url": "https://arxiv.org/pdf/2606.25115", + "primary_query": "agent-memory" + }, + { + "id": "2606.24322", + "title": "Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees", + "url": "https://arxiv.org/abs/2606.24322", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yedidel Louck" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.24322", + "source": "arxiv", + "source_id": "arxiv:2606.24322", + "pdf_url": "https://arxiv.org/pdf/2606.24322", + "primary_query": "agent-memory" + }, + { + "id": "2606.24839", + "title": "Grading the Grader: Lessons from Evaluating an Agentic Data Analysis System", + "url": "https://arxiv.org/abs/2606.24839", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Tian Zheng", + "Kai-Tai Hsu" + ], + "categories": [ + "cs.AI", + "stat.AP" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24839", + "source": "arxiv", + "source_id": "arxiv:2606.24839", + "pdf_url": "https://arxiv.org/pdf/2606.24839", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.23927", + "title": "RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems", + "url": "https://arxiv.org/abs/2606.23927", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Yarin Yerushalmi Levi", + "Roy Betser", + "Amit Giloni", + "Lidor Erez", + "Itay Gershon", + "Oren Rachmil", + "Sindhu Padakandla", + "Roman Vainshtein" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.23927", + "source": "arxiv", + "source_id": "arxiv:2606.23927", + "pdf_url": "https://arxiv.org/pdf/2606.23927", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24551", + "title": "GUI vs. CLI: Execution Bottlenecks in Screen-Only and Skill-Mediated Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.24551", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Xiao Zhou", + "Siyue Zhang", + "Yilun Zhao", + "Jinbiao Wei", + "Tingyu Song", + "Arman Cohan", + "Chen Zhao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.24551", + "source": "arxiv", + "source_id": "arxiv:2606.24551", + "pdf_url": "https://arxiv.org/pdf/2606.24551", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.22948", + "title": "ENVS: Environment-Native Verified Search for Long-Horizon GUI Agents", + "url": "https://arxiv.org/abs/2606.22948", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Yincheng Zhou", + "Athena Zhuoming Zhong", + "Shijie Zhang", + "Kevin Zhang", + "Teresa Xiaotao Shang", + "Shanghang Zhang" + ], + "categories": [ + "cs.AI", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.22948", + "source": "arxiv", + "source_id": "arxiv:2606.22948", + "pdf_url": "https://arxiv.org/pdf/2606.22948", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.23764", + "title": "Emergent Relational Order in LLM Agent Societies: From Collective Affect to Authority Stratification", + "url": "https://arxiv.org/abs/2606.23764", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Zhiyuan Ji", + "Xinyu Chen", + "Ziqi Dai", + "Shiyun Tang", + "Chunyu Wei", + "Yueguo Chen" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.23764", + "source": "arxiv", + "source_id": "arxiv:2606.23764", + "pdf_url": "https://arxiv.org/pdf/2606.23764", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.23343", + "title": "Group Selection Promotes Prosocial Prompts in Populations of LLM Agents", + "url": "https://arxiv.org/abs/2606.23343", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Luis Celiktemel", + "Edward Eichhorn", + "Levin Brinkmann", + "Robin Schimmelpfennig", + "Aron Vallinder", + "Yaomin Jiang", + "Edward Hughes", + "Iyad Rahwan" + ], + "categories": [ + "cs.CY" + ], + "topics": [ + "coding-agent", + "computer-use", + "multi-agent", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.23343", + "source": "arxiv", + "source_id": "arxiv:2606.23343", + "pdf_url": "https://arxiv.org/pdf/2606.23343", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22388", + "title": "PlanBench-XL: Evaluating Long-Horizon Planning of LLM Tool-Use Agents in Large-Scale Tool Ecosystems", + "url": "https://arxiv.org/abs/2606.22388", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Jiayu Liu", + "Qihan Lin", + "Cheng Qian", + "Rui Wang", + "Emre Can Acikgoz", + "Xiaocheng Yang", + "Jiateng Liu", + "Zhenhailong Wang", + "Xiusi Chen", + "Heng Ji", + "Dilek Hakkani-Tür" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent", + "tool-use" + ], + "arxiv_id": "2606.22388", + "source": "arxiv", + "source_id": "arxiv:2606.22388", + "pdf_url": "https://arxiv.org/pdf/2606.22388", + "primary_query": "planning-agent" + }, + { + "id": "2606.22610", + "title": "PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement", + "url": "https://arxiv.org/abs/2606.22610", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Weiwei Ye", + "Hangchen Liu", + "Dongyuan Li", + "Renhe Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.22610", + "source": "arxiv", + "source_id": "arxiv:2606.22610", + "pdf_url": "https://arxiv.org/pdf/2606.22610", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.21565", + "title": "Composing Verifiable Conceptual Models via Building Blocks: Towards Design-Time Verification of Agentic AI Workflows", + "url": "https://arxiv.org/abs/2606.21565", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Noe Y. Flandre", + "Alexander C. Nwala", + "Philippe J. Giabbanelli" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.21565", + "source": "arxiv", + "source_id": "arxiv:2606.21565", + "pdf_url": "https://arxiv.org/pdf/2606.21565", + "primary_query": "agentic-ai" + }, + { + "id": "2606.21732", + "title": "Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents", + "url": "https://arxiv.org/abs/2606.21732", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Zesen Liu", + "Zihan Zhang", + "Dongdong She" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.21732", + "source": "arxiv", + "source_id": "arxiv:2606.21732", + "pdf_url": "https://arxiv.org/pdf/2606.21732", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.20479", + "title": "GroundControl: Anticipating Navigation Failures in Vision-Language Agents via Trajectory-Consistent Uncertainty Estimates", + "url": "https://arxiv.org/abs/2606.20479", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Nastaran Darabi", + "Divake Kumar", + "Sina Tayebati", + "Devashri Naik", + "Amit Ranjan Trivedi" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.20479", + "source": "arxiv", + "source_id": "arxiv:2606.20479", + "pdf_url": "https://arxiv.org/pdf/2606.20479", + "primary_query": "language-agent" + }, + { + "id": "2606.20041", + "title": "AI Economist Agent: An Agentic Framework for Model-Grounded Economic Analysis with RAG, Knowledge Graphs, and Large Language Models", + "url": "https://arxiv.org/abs/2606.20041", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Masahiro Kato" + ], + "categories": [ + "econ.GN", + "cs.AI", + "cs.LG", + "q-fin.GN" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2606.20041", + "source": "arxiv", + "source_id": "arxiv:2606.20041", + "pdf_url": "https://arxiv.org/pdf/2606.20041", + "primary_query": "ai-agent" + }, + { + "id": "2606.19852", + "title": "Prompt, Plan, Extract: Zero-Shot Agentic LLMs Workflows for Lung Pathology Extraction from Clinical Narratives", + "url": "https://arxiv.org/abs/2606.19852", + "published": "2026-06-18", + "updated": "2026-06-25", + "authors": [ + "Aman Pathak", + "Cheng Peng", + "Mengxian Lyu", + "Ziyi Chen", + "Reema Solan", + "Sankalp Talankar", + "Yasir Khan", + "Hiren Mehta", + "Aokun Chen", + "Yi Guo", + "Yonghui Wu" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.19852", + "source": "arxiv", + "source_id": "arxiv:2606.19852", + "pdf_url": "https://arxiv.org/pdf/2606.19852", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20047", + "title": "PACMS: Submodular Context Selection as a Pluggable Engine for LLM Agents", + "url": "https://arxiv.org/abs/2606.20047", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Manu Ghulyani", + "Arunabh Singh", + "Karan Bharadwaj", + "Ankit Nath", + "Suranjan Goswami" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.20047", + "source": "arxiv", + "source_id": "arxiv:2606.20047", + "pdf_url": "https://arxiv.org/pdf/2606.20047", + "primary_query": "tool-use" + }, + { + "id": "2606.19926", + "title": "MemGUI-Agent: An End-to-End Long-Horizon Mobile GUI Agent with Proactive Context Management", + "url": "https://arxiv.org/abs/2606.19926", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Guangyi Liu", + "Gao Wu", + "Congxiao Liu", + "Pengxiang Zhao", + "Liang Liu", + "Mading Li", + "Qi Zhang", + "Mengyan Wang", + "Liang Guo", + "Yong Liu" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.19926", + "source": "arxiv", + "source_id": "arxiv:2606.19926", + "pdf_url": "https://arxiv.org/pdf/2606.19926", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.20922", + "title": "Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning", + "url": "https://arxiv.org/abs/2606.20922", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Shanghao Shi", + "Xiao Wang", + "Chaoyu Zhang", + "Hao Li", + "Wenjing Lou", + "Thomas Hou", + "Yevgeniy Vorobeychik", + "Chongjie Zhang", + "Ning Zhang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.20922", + "source": "arxiv", + "source_id": "arxiv:2606.20922", + "pdf_url": "https://arxiv.org/pdf/2606.20922", + "primary_query": "planning-agent" + }, + { + "id": "2606.19787", + "title": "ORAgentBench: Can LLM Agents Solve Challenging Operations Research Tasks End to End?", + "url": "https://arxiv.org/abs/2606.19787", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Jiajun Li", + "Mingshu Cai", + "Yixuan Li", + "Yu Ding", + "Ran Hou", + "Guanyu Nie", + "Xiongwei Han", + "Wanyuan Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.19787", + "source": "arxiv", + "source_id": "arxiv:2606.19787", + "pdf_url": "https://arxiv.org/pdf/2606.19787", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.18068", + "title": "Agentic AI-based Framework for Mitigating Premature Diagnostic Handoff and Silent Hallucination in Healthcare Applications", + "url": "https://arxiv.org/abs/2606.18068", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Divyansh Srivastava", + "Shreya Ghosh", + "Anshul Verma", + "Rajkumar Buyya" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.18068", + "source": "arxiv", + "source_id": "arxiv:2606.18068", + "pdf_url": "https://arxiv.org/pdf/2606.18068", + "primary_query": "agentic-ai" + }, + { + "id": "2606.17680", + "title": "EnvRL: Learn from Environment Dynamics in Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.17680", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Zhitong Wang", + "Songze Li", + "Hao Peng", + "Shuzheng Si", + "Yi Wang", + "Maosong Sun", + "Juanzi Li" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.17680", + "source": "arxiv", + "source_id": "arxiv:2606.17680", + "pdf_url": "https://arxiv.org/pdf/2606.17680", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18406", + "title": "CoreMem: Riemannian Retrieval and Fisher-Guided Distillation for Long-Term Memory in Dialogue Agents", + "url": "https://arxiv.org/abs/2606.18406", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Jiaqi Chen", + "Yongqin Zeng", + "Shaoshen Chen", + "Yijian Zhang", + "Hai-Tao Zheng", + "Chunxia Ma", + "XiuTeng Zhou" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.18406", + "source": "arxiv", + "source_id": "arxiv:2606.18406", + "pdf_url": "https://arxiv.org/pdf/2606.18406", + "primary_query": "agent-memory" + }, + { + "id": "2606.17449", + "title": "MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation", + "url": "https://arxiv.org/abs/2606.17449", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Zehang Wei", + "Jiaxin Dai", + "Jiamin Yan", + "Xiang Xiang" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.CV", + "cs.LG", + "cs.MM" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.17449", + "source": "arxiv", + "source_id": "arxiv:2606.17449", + "pdf_url": "https://arxiv.org/pdf/2606.17449", + "primary_query": "rag-agent" + }, + { + "id": "2606.16432", + "title": "ACCORD: Action-Conditioned Contextual Grounding for Language Agents", + "url": "https://arxiv.org/abs/2606.16432", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Lai Jiang", + "Cheng Qian", + "Zhenhailong Wang", + "Pan Lu", + "Heng Ji", + "Hao Peng" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.16432", + "source": "arxiv", + "source_id": "arxiv:2606.16432", + "pdf_url": "https://arxiv.org/pdf/2606.16432", + "primary_query": "language-agent" + }, + { + "id": "2606.15903", + "title": "Control-Plane Placement Shapes Forgetting: An Architectural Study of Agent Memory Across Thirteen System Configurations", + "url": "https://arxiv.org/abs/2606.15903", + "published": "2026-06-14", + "updated": "2026-06-16", + "authors": [ + "Dongxu Yang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.15903", + "source": "arxiv", + "source_id": "arxiv:2606.15903", + "pdf_url": "https://arxiv.org/pdf/2606.15903", + "primary_query": "agent-memory" + }, + { + "id": "2606.15609", + "title": "FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion", + "url": "https://arxiv.org/abs/2606.15609", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Zixin Rao", + "Wentian Zhu", + "Chan Aristella Lu", + "Zhaorun Chen", + "Wei Niu", + "Le Guan", + "Bo Li", + "Zhen Xiang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.15609", + "source": "arxiv", + "source_id": "arxiv:2606.15609", + "pdf_url": "https://arxiv.org/pdf/2606.15609", + "primary_query": "agent-memory" + }, + { + "id": "2606.15079", + "title": "Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale", + "url": "https://arxiv.org/abs/2606.15079", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Ang Li", + "Ben Liu", + "Bin Han", + "Bin Hu", + "Bin Jing", + "Binbin Hu", + "Bing Li", + "Cai Chen", + "Caizhi Tang", + "Changxin Tian", + "Chao Huang", + "Chao Zhang", + "Chen Liang", + "Chen Qian", + "Chengfu Tang", + "Chengyao Wen", + "Chilin Fu", + "Chunwei Wu", + "Cong Zhang", + "Cunyin Peng", + "Daixin Wang", + "Dalong Zhang", + "Deng Zhao", + "Dingnan Jin", + "Dingyuan Zhu", + "Donghao Zhang", + "Fan Yuan", + "Fangzheng Zhao", + "Fanzhuang Meng", + "Feifan Wu", + "Feng Xu", + "Fengbin Fang", + "Gangshan Wang", + "Guodong Yang", + "Hailin Zhao", + "Haitao Wang", + "Haitao Zhang", + "Hanxiao Zhang", + "Hanzi Wang", + "Hao Dai", + "Hao Liu", + "Hao Qian", + "Hao Wu", + "Haoxiong Liu", + "Haoyu Xu", + "Heng Zhang", + "Hong Liu", + "Hongliang Zhang", + "Hongrui Liu", + "Hongxun Li", + "Hongzhi Ruan", + "Huaidong Xiong", + "Huihuang Zheng", + "Huikang Tang", + "Jia Guo", + "Jia Li", + "Jia Liu", + "Jiameng Wang", + "Jiaming Liu", + "Jiannan Shi", + "Jianping Wei", + "Jiaolong Yang", + "Jiapeng Wang", + "Jie Gao", + "Jie Wang", + "Jiewei Wu", + "Jin Yang", + "Jinjin Li", + "Jinjing Huang", + "Jinquan Sun", + "Jinyao Chen", + "Juanhui Tu", + "Jun Liu", + "Jun Mei", + "Jun Xu", + "Jun Zhou", + "Junjie Ou", + "Junnan Sipan", + "Junpeng Fang", + "Kaihong Zhang", + "Kaiqin Hu", + "Ke Shi", + "Kuan Xu", + "Kun Tang", + "Kunlong Chen", + "Lanyin Mei", + "Lei Chen", + "Lei Liang", + "Lei Xu", + "Li Tang", + "Liang Jiang", + "Liangcheng Fu", + "Lihui Zhang", + "Linfeng Shi", + "Lintao Ma", + "Liyuan Liu", + "Longfei Li", + "Longfei Zheng", + "Lu Liu", + "Lu Yu", + "Man Li", + "Meiqi Zhu", + "Meng Li", + "Mengjie Gao", + "Mengshu Sun", + "Mingming Yin", + "Mingyang Zhang", + "Mingyuan Fan", + "Nuo Xu", + "Pan Tang", + "Peijie Jiang", + "Peilong Zhao", + "Peng Lin", + "Pingping Liu", + "Qi Zuo", + "Qian Zhao", + "Qiang Cheng", + "Qianggang Cao", + "Qiaoben Bao", + "Qing Cui", + "Qingyuan Yang", + "Qitao Shi", + "Qiyin Huang", + "Qizheng Zhou", + "Quan Wan", + "Runyuan Zhao", + "Shaomian Zheng", + "Shaowei Wei", + "Shengnan Zhang", + "Shuaicheng Li", + "Shujie Li", + "Shuo Zhang", + "Sikang Bian", + "Tianchu Yao", + "Tiange Xu", + "Tianshu Wang", + "Ting Guo", + "Tinghao Wang", + "Tingwei Huang", + "Tong Zhao", + "Tongkai Yang", + "Wang Hong", + "Wanli Gu", + "Wei Lu", + "Weichang Wu", + "Weiguang Han", + "Weiquan Li", + "Wenbo Shen", + "Wenjing Fang", + "Wenzhi Tang", + "Xiang Shu", + "Xiao Shi", + "Xiaodong Yan", + "Xiaolu Zhang", + "Xiaopei Wan", + "Xiaqing Sun", + "Xin Zhao", + "Xingyu Lu", + "Xinxing Yang", + "Xinyao Tang", + "Xinyu Kong", + "Xinyu Liu", + "Xiong Xu", + "Xuan Sun", + "Xudong Han", + "Xudong Wang", + "Xujie Shen", + "Yalin Zhang", + "Yangyang Hou", + "Yankun Ren", + "Yao Zhao", + "Ye Chen", + "Yeyang Chen", + "Yibo Cao", + "Yifan Zuo", + "Yijie Chen", + "Ying Li", + "Yingjie Song", + "Yingxue Li", + "Yiqi Wang", + "Yixuan Sun", + "Yizhu Xiao", + "Yongfei Xu", + "Yu Liu", + "Yuchen Fang", + "Yue Gao", + "Yue Yu", + "Yue Zhang", + "Yuqi Zhang", + "Yuxiao He", + "Yuxiao Lu", + "Yuxin Tian", + "Yuxuan Li", + "Yuzhuo Fu", + "Zhankai Xu", + "Zhaoxin Huan", + "Zhenduo Zhang", + "Zhengke Gui", + "Zhengyu Huang", + "Zhenjun Ma", + "Zhenxuan Pan", + "Zheping Qu", + "Zhibo Zhu", + "Zhidong Fan", + "Zhigang Huangfu", + "Zhihao Wang", + "Zhiqiang Zhang", + "Zhizhen Liu", + "Zhuyan Zhou", + "Zibin Lin", + "Zihang Zeng", + "Zihao Wang", + "Zilong Wang", + "Ziqi Liu", + "Zitao Xuan", + "Zixuan Cheng", + "Zujie Wen", + "Zuoli Tang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.15079", + "source": "arxiv", + "source_id": "arxiv:2606.15079", + "pdf_url": "https://arxiv.org/pdf/2606.15079", + "primary_query": "tool-use" + }, + { + "id": "2606.14571", + "title": "StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance", + "url": "https://arxiv.org/abs/2606.14571", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Guanming Liu", + "Yuqi Ren", + "Hansu Gu", + "Peng Zhang", + "Weihang Wang", + "Jiahao Liu", + "Ning Gu", + "Tun Lu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.14571", + "source": "arxiv", + "source_id": "arxiv:2606.14571", + "pdf_url": "https://arxiv.org/pdf/2606.14571", + "primary_query": "agent-memory" + }, + { + "id": "2606.13643", + "title": "Recursive Agent Harnesses", + "url": "https://arxiv.org/abs/2606.13643", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Elias Lumer", + "Sahil Sen", + "Kevin Paul", + "Vamse Kumar Subbiah" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2606.13643", + "source": "arxiv", + "source_id": "arxiv:2606.13643", + "pdf_url": "https://arxiv.org/pdf/2606.13643", + "primary_query": "function-calling" + }, + { + "id": "2606.13385", + "title": "Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents", + "url": "https://arxiv.org/abs/2606.13385", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Zihao Wang", + "Yiming Li", + "Yutong Wu", + "Zheyu Liu", + "Kangjie Chen", + "Fok Kar Wai", + "Pin-Yu Chen", + "Vrizlynn L. L. Thing", + "Bo Li", + "Dacheng Tao", + "Tianwei Zhang" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CY", + "cs.HC", + "cs.MM" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.13385", + "source": "arxiv", + "source_id": "arxiv:2606.13385", + "pdf_url": "https://arxiv.org/pdf/2606.13385", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13904", + "title": "SANA: What Matters for QA Agents over Massive Data Lakes?", + "url": "https://arxiv.org/abs/2606.13904", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Austin Senna Wijaya", + "Jiaxiang Liu", + "Haonan Wang", + "Eugene Wu" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.DB" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.13904", + "source": "arxiv", + "source_id": "arxiv:2606.13904", + "pdf_url": "https://arxiv.org/pdf/2606.13904", + "primary_query": "planning-agent" + }, + { + "id": "2606.12341", + "title": "OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents", + "url": "https://arxiv.org/abs/2606.12341", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Jin Xie", + "Songze Li" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.12341", + "source": "arxiv", + "source_id": "arxiv:2606.12341", + "pdf_url": "https://arxiv.org/pdf/2606.12341", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.12195", + "title": "InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning", + "url": "https://arxiv.org/abs/2606.12195", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Ziang Yan", + "Sheng Xia", + "Jiashuo Yu", + "Yue Wu", + "Tianxiang Jiang", + "Songze Li", + "Kanghui Tian", + "Yicheng Xu", + "Yinan He", + "Kai Chen", + "Limin Wang", + "Yu Qiao", + "Yi Wang" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.12195", + "source": "arxiv", + "source_id": "arxiv:2606.12195", + "pdf_url": "https://arxiv.org/pdf/2606.12195", + "primary_query": "tool-use" + }, + { + "id": "2606.11869", + "title": "Agents All the Way Down; A Methodology for Building Custom AI Agents from Substrate to Production", + "url": "https://arxiv.org/abs/2606.11869", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Marc Alier Forment", + "Juanan Pereira", + "Francisco José García-Peñalvo", + "María José Casañ Guerrero" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2606.11869", + "source": "arxiv", + "source_id": "arxiv:2606.11869", + "pdf_url": "https://arxiv.org/pdf/2606.11869", + "primary_query": "function-calling" + }, + { + "id": "2606.17076", + "title": "CMIP-Forge: An Agentic System that Retrieves, Computes, and Self-Reviews Climate Science", + "url": "https://arxiv.org/abs/2606.17076", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Dmitrii Pantiukhin", + "Boris Shapkin", + "Ivan Kuznetsov", + "Thomas Jung", + "Nikolay Koldunov" + ], + "categories": [ + "physics.ao-ph", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.17076", + "source": "arxiv", + "source_id": "arxiv:2606.17076", + "pdf_url": "https://arxiv.org/pdf/2606.17076", + "primary_query": "rag-agent" + }, + { + "id": "2606.11349", + "title": "Knowing When to Ask: Self-Gated Clarification for Hierarchical Language Agents", + "url": "https://arxiv.org/abs/2606.11349", + "published": "2026-06-09", + "updated": "2026-06-12", + "authors": [ + "Aijing Gao", + "Yiming Kang", + "Mengdie Flora Wang", + "Jae Oh Woo" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.11349", + "source": "arxiv", + "source_id": "arxiv:2606.11349", + "pdf_url": "https://arxiv.org/pdf/2606.11349", + "primary_query": "language-agent" + }, + { + "id": "2606.11078", + "title": "A History-Aware Visually Grounded Critic for Computer Use Agents", + "url": "https://arxiv.org/abs/2606.11078", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Jaewoo Lee", + "Zaid Khan", + "Archiki Prasad", + "Justin Chih-Yao Chen", + "Supriyo Chakraborty", + "Kartik Balasubramaniam", + "Sambit Sahu", + "Elias Stengel-Eskin", + "Hyunji Lee", + "Mohit Bansal" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.11078", + "source": "arxiv", + "source_id": "arxiv:2606.11078", + "pdf_url": "https://arxiv.org/pdf/2606.11078", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.10423", + "title": "WebChallenger: A Reliable and Efficient Generalist Web Agent", + "url": "https://arxiv.org/abs/2606.10423", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Jayoo Hwang", + "Xiaowen Zhang", + "Vedant Padwal" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "computer-use", + "embodied-agent", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.10423", + "source": "arxiv", + "source_id": "arxiv:2606.10423", + "pdf_url": "https://arxiv.org/pdf/2606.10423", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.10381", + "title": "Agentic Hybrid RAG for Evidence-Grounded Muon Collider Analysis", + "url": "https://arxiv.org/abs/2606.10381", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Ruobing Jiang", + "Dawei Fu", + "Cheng Jiang", + "Tianyi Yang", + "Zijian Wang", + "Youpeng Wu", + "Yong Ban", + "Yajun Mao", + "Qiang Li" + ], + "categories": [ + "hep-ex", + "cs.AI", + "cs.CL", + "cs.IR", + "physics.ins-det" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.10381", + "source": "arxiv", + "source_id": "arxiv:2606.10381", + "pdf_url": "https://arxiv.org/pdf/2606.10381", + "primary_query": "rag-agent" + }, + { + "id": "2606.09764", + "title": "iOSWorld: A Benchmark for Personally Intelligent Phone Agents", + "url": "https://arxiv.org/abs/2606.09764", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Lawrence Keunho Jang", + "Mareks Woodside", + "Geronimo Carom", + "Andrew Keunwoo Jang", + "Jing Yu Koh", + "Ruslan Salakhutdinov" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "web-gui-agent" + ], + "arxiv_id": "2606.09764", + "source": "arxiv", + "source_id": "arxiv:2606.09764", + "pdf_url": "https://arxiv.org/pdf/2606.09764", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.09399", + "title": "RunAgent SuperBrowser: A Theory of Autonomous Web Navigation Grounded in Human Browsing Behaviour", + "url": "https://arxiv.org/abs/2606.09399", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Radeen Mostafa", + "Sawradip Saha" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.09399", + "source": "arxiv", + "source_id": "arxiv:2606.09399", + "pdf_url": "https://arxiv.org/pdf/2606.09399", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09549", + "title": "SecureClaw: Clawing Back Control of LLM Agents", + "url": "https://arxiv.org/abs/2606.09549", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Yuhan Ma", + "Stefan Schmid" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "planning-agent" + ], + "arxiv_id": "2606.09549", + "source": "arxiv", + "source_id": "arxiv:2606.09549", + "pdf_url": "https://arxiv.org/pdf/2606.09549", + "primary_query": "agent-safety" + }, + { + "id": "2606.09071", + "title": "REFLECT: Intervention-Supported Error Attribution for Silent Failures in LLM Agent Traces", + "url": "https://arxiv.org/abs/2606.09071", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Xiaofeng Lin", + "Yingxu Wang", + "Tung Sum Thomas Kwok", + "Daniel Guo", + "Sahil Arun Nale", + "Charles Fleming", + "Guang Cheng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.09071", + "source": "arxiv", + "source_id": "arxiv:2606.09071", + "pdf_url": "https://arxiv.org/pdf/2606.09071", + "primary_query": "planning-agent" + }, + { + "id": "2606.05463", + "title": "PSEBench: A Controllable and Verifiable Benchmark for Evaluating LLMs in Patient Safety Event Triage", + "url": "https://arxiv.org/abs/2606.05463", + "published": "2026-06-03", + "updated": "2026-06-09", + "authors": [ + "Keqi Han", + "Ryan Young", + "Annabel Strauss", + "Lindsey Hughes", + "Katharine M. Nesbitt", + "Nicole Schueler", + "Che Ngufor", + "Carl Yang", + "Yuan Xue", + "Zhijun Yin" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.05463", + "source": "arxiv", + "source_id": "arxiv:2606.05463", + "pdf_url": "https://arxiv.org/pdf/2606.05463", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.03135", + "title": "Uncertainty-Aware Clarification in LLM Agents with Information Gain", + "url": "https://arxiv.org/abs/2606.03135", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Mengyi Deng", + "Zhiwei Li", + "Xin Li", + "Tingyu Zhu", + "Ying Zhao", + "Zhijiang Guo", + "Wei Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.03135", + "source": "arxiv", + "source_id": "arxiv:2606.03135", + "pdf_url": "https://arxiv.org/pdf/2606.03135", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.04120", + "title": "SaliMory: Orchestrating Cognitive Memory for Conversational Agents", + "url": "https://arxiv.org/abs/2606.04120", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Kai Zhang", + "Xinyuan Zhang", + "Hongda Jiang", + "Shiun-Zu Kuo", + "Hyokun Yun", + "Ejaz Ahmed", + "Shereen Oraby", + "Ziyun Li", + "Sanat Sharma", + "Ann Lee", + "Ahmed A Aly", + "Anuj Kumar", + "Raffay Hamid", + "Xin Luna Dong" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04120", + "source": "arxiv", + "source_id": "arxiv:2606.04120", + "pdf_url": "https://arxiv.org/pdf/2606.04120", + "primary_query": "agent-memory" + }, + { + "id": "2606.04296", + "title": "The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents", + "url": "https://arxiv.org/abs/2606.04296", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Manvendra Modgil" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.04296", + "source": "arxiv", + "source_id": "arxiv:2606.04296", + "pdf_url": "https://arxiv.org/pdf/2606.04296", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.03108", + "title": "EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.03108", + "published": "2026-06-02", + "updated": "2026-06-12", + "authors": [ + "Guhong Chen", + "Yingcheng Shi", + "Yongbin Li", + "Binhua Li", + "Xander Xu", + "Hu Wei", + "Shiwen Ni", + "Min Yang", + "Jieping Ye" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.03108", + "source": "arxiv", + "source_id": "arxiv:2606.03108", + "pdf_url": "https://arxiv.org/pdf/2606.03108", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.02388", + "title": "Policy and World Modeling Co-Training for Language Agents", + "url": "https://arxiv.org/abs/2606.02388", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Ning Lu", + "Baijiong Lin", + "Shengcai Liu", + "Jiahao Wu", + "Haoze Lv", + "Yanbin Wei", + "Lingting Zhu", + "Shengju Qian", + "Xin Wang", + "Ying-Cong Chen", + "Qi Wang", + "Ke Tang" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.02388", + "source": "arxiv", + "source_id": "arxiv:2606.02388", + "pdf_url": "https://arxiv.org/pdf/2606.02388", + "primary_query": "language-agent" + }, + { + "id": "2606.01815", + "title": "CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation", + "url": "https://arxiv.org/abs/2606.01815", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Danqing Wang", + "Akshay Sivaraman", + "Lei Li" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.01815", + "source": "arxiv", + "source_id": "arxiv:2606.01815", + "pdf_url": "https://arxiv.org/pdf/2606.01815", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.02380", + "title": "SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence", + "url": "https://arxiv.org/abs/2606.02380", + "published": "2026-06-01", + "updated": "2026-06-28", + "authors": [ + "Yuyan Bu", + "Haowei Li", + "Qirui Zheng", + "Bowen Dong", + "Kaiyue Yang", + "Jiaming Ji", + "Yingshui Tan", + "Wenxin Li", + "Yaodong Yang", + "Juntao Dai" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.02380", + "source": "arxiv", + "source_id": "arxiv:2606.02380", + "pdf_url": "https://arxiv.org/pdf/2606.02380", + "primary_query": "agent-safety" + }, + { + "id": "2606.00914", + "title": "Adversarial Feeds Steer LLM Agent Decisions Against Their Defaults", + "url": "https://arxiv.org/abs/2606.00914", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Rana Muhammad Usman" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.00914", + "source": "arxiv", + "source_id": "arxiv:2606.00914", + "pdf_url": "https://arxiv.org/pdf/2606.00914", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.00915", + "title": "Autonomous agentic design for photonics", + "url": "https://arxiv.org/abs/2606.00915", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Prashanta Kharel", + "Amin Khavasi", + "Xinzhong Chen", + "Tyler W. Hughes" + ], + "categories": [ + "physics.optics" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.00915", + "source": "arxiv", + "source_id": "arxiv:2606.00915", + "pdf_url": "https://arxiv.org/pdf/2606.00915", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.31278", + "title": "Industrializing Prediction-Powered Inference: The GLIDE Library for Reliable GenAI and Agentic Systems Evaluation", + "url": "https://arxiv.org/abs/2605.31278", + "published": "2026-05-29", + "updated": "2026-06-04", + "authors": [ + "Grégoire Martinon", + "Ibrahim Merad", + "Mohammed Raki" + ], + "categories": [ + "cs.AI", + "cs.LG", + "stat.ME" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.31278", + "source": "arxiv", + "source_id": "arxiv:2605.31278", + "pdf_url": "https://arxiv.org/pdf/2605.31278", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.29676", + "title": "Notation Matters: A Benchmark Study of Token-Optimized Formats in Agentic AI Systems", + "url": "https://arxiv.org/abs/2605.29676", + "published": "2026-05-28", + "updated": "2026-06-17", + "authors": [ + "Lorenz Kutschka", + "Bernhard Geiger" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.29676", + "source": "arxiv", + "source_id": "arxiv:2605.29676", + "pdf_url": "https://arxiv.org/pdf/2605.29676", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.28046", + "title": "MemCog: From Memory-as-Tool to Memory-as-Cognition in Conversational Agents", + "url": "https://arxiv.org/abs/2605.28046", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Zihan Li", + "Xingyu Fan", + "Feifei Li", + "Wenhui Que" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.28046", + "source": "arxiv", + "source_id": "arxiv:2605.28046", + "pdf_url": "https://arxiv.org/pdf/2605.28046", + "primary_query": "agent-memory" + }, + { + "id": "2605.28607", + "title": "Adaptive Multimodal Agents-Based Framework for Automatic Workflow Execution", + "url": "https://arxiv.org/abs/2605.28607", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Susanna Cifani", + "Mario Luca Bernardi", + "Marta Cimitile" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "multi-agent", + "planning", + "rag", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.28607", + "source": "arxiv", + "source_id": "arxiv:2605.28607", + "pdf_url": "https://arxiv.org/pdf/2605.28607", + "primary_query": "rag-agent" + }, + { + "id": "2605.28120", + "title": "LegalGraphRAG: Multi-Agent Graph Retrieval-Augmented Generation for Reliable Legal Reasoning", + "url": "https://arxiv.org/abs/2605.28120", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Zerui Chen", + "Qinggang Zhang", + "Zhishang Xiang", + "Zhimin Wei", + "Linfeng Gao", + "Xiao Huang", + "Zhihong Zhang", + "Jinsong Su" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.28120", + "source": "arxiv", + "source_id": "arxiv:2605.28120", + "pdf_url": "https://arxiv.org/pdf/2605.28120", + "primary_query": "rag-agent" + }, + { + "id": "2605.28787", + "title": "Do Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval", + "url": "https://arxiv.org/abs/2605.28787", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Shiyu Chen", + "Tarfah Alrashed", + "Alon Halevy", + "Natasha Noy" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.28787", + "source": "arxiv", + "source_id": "arxiv:2605.28787", + "pdf_url": "https://arxiv.org/pdf/2605.28787", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.27366", + "title": "MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation", + "url": "https://arxiv.org/abs/2605.27366", + "published": "2026-05-26", + "updated": "2026-07-03", + "authors": [ + "Huawei Lin", + "Peng Li", + "Jie Song", + "Fuxin Jiang", + "Tieying Zhang" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.27366", + "source": "arxiv", + "source_id": "arxiv:2605.27366", + "pdf_url": "https://arxiv.org/pdf/2605.27366", + "primary_query": "agent-memory" + }, + { + "id": "2605.26926", + "title": "From Norms to Indicators (N2I-RAG): An Agentic Retrieval-Augmented Generation Framework for Legal Indicator Computation", + "url": "https://arxiv.org/abs/2605.26926", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Youssef Al Mouatamid", + "Marie Bonnin", + "Jihad Zahir" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.26926", + "source": "arxiv", + "source_id": "arxiv:2605.26926", + "pdf_url": "https://arxiv.org/pdf/2605.26926", + "primary_query": "rag-agent" + }, + { + "id": "2605.27333", + "title": "FinHarness: An Inline Lifecycle Safety Harness for Finance LLM Agents", + "url": "https://arxiv.org/abs/2605.27333", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Haoxuan Jia", + "Yang Liu", + "Bin Chong", + "Yingguang Yang", + "Yancheng Chen", + "Jiayu Liang", + "Qian Li", + "Hanning Lu", + "Kefu Xu", + "Hao Zheng", + "Chongyang Zhang", + "Hao Peng", + "Philip S. Yu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.27333", + "source": "arxiv", + "source_id": "arxiv:2605.27333", + "pdf_url": "https://arxiv.org/pdf/2605.27333", + "primary_query": "planning-agent" + }, + { + "id": "2605.26720", + "title": "Towards Feedback-to-Plan Decisions for Self-Evolving LLM Agents in CUDA Kernel Generation", + "url": "https://arxiv.org/abs/2605.26720", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Yee Hin Chong", + "Jiaming Wu", + "Youhui Zhang", + "Peng Qu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.26720", + "source": "arxiv", + "source_id": "arxiv:2605.26720", + "pdf_url": "https://arxiv.org/pdf/2605.26720", + "primary_query": "planning-agent" + }, + { + "id": "2605.26252", + "title": "Is Agent Memory a Database? Rethinking Data Foundations for Long-Term AI Agent Memory", + "url": "https://arxiv.org/abs/2605.26252", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Abdelghny Orogat", + "Essam Mansour" + ], + "categories": [ + "cs.AI", + "cs.DB" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.26252", + "source": "arxiv", + "source_id": "arxiv:2605.26252", + "pdf_url": "https://arxiv.org/pdf/2605.26252", + "primary_query": "agent-memory" + }, + { + "id": "2605.26305", + "title": "Experiments in Agentic AI for Science", + "url": "https://arxiv.org/abs/2605.26305", + "published": "2026-05-25", + "updated": "2026-05-29", + "authors": [ + "Judy Fox", + "Geoffrey Fox" + ], + "categories": [ + "cs.AI", + "eess.SY", + "hep-ph" + ], + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "rag-agent" + ], + "arxiv_id": "2605.26305", + "source": "arxiv", + "source_id": "arxiv:2605.26305", + "pdf_url": "https://arxiv.org/pdf/2605.26305", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.23636", + "title": "RF Instrument Agent (RFIA): Empowering RF Instruments with Natural Language Understanding, Scheduling and Execution of Complex Tasks", + "url": "https://arxiv.org/abs/2605.23636", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Chunhui Li", + "Wei Fan" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.23636", + "source": "arxiv", + "source_id": "arxiv:2605.23636", + "pdf_url": "https://arxiv.org/pdf/2605.23636", + "primary_query": "language-agent" + }, + { + "id": "2605.22321", + "title": "Benchmarking Autonomous Agents against Temporal, Spatial, and Semantic Evasions", + "url": "https://arxiv.org/abs/2605.22321", + "published": "2026-05-21", + "updated": "2026-05-21", + "authors": [ + "Jianan Ma", + "Xiaohu Du", + "Ruixiao Lin", + "Yaoxiang Bian", + "Jialuo Chen", + "Jingyi Wang", + "Xiaofang Yang", + "Shiwen Cui", + "Changhua Meng", + "Xinhao Deng", + "Zhen Wang" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.22321", + "source": "arxiv", + "source_id": "arxiv:2605.22321", + "pdf_url": "https://arxiv.org/pdf/2605.22321", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.21740", + "title": "SMDD-Bench: Can LLMs Solve Real-World Small Molecule Drug Design Tasks?", + "url": "https://arxiv.org/abs/2605.21740", + "published": "2026-05-20", + "updated": "2026-05-24", + "authors": [ + "Kevin Han", + "Renfei Zhang", + "Kathy Wei", + "Hamed Mahdavi", + "Niloofar Mireshghallah", + "Amir Barati Farimani" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.21740", + "source": "arxiv", + "source_id": "arxiv:2605.21740", + "pdf_url": "https://arxiv.org/pdf/2605.21740", + "primary_query": "planning-agent" + }, + { + "id": "2605.20874", + "title": "Governance by Construction for Generalist Agents", + "url": "https://arxiv.org/abs/2605.20874", + "published": "2026-05-20", + "updated": "2026-05-20", + "authors": [ + "Segev Shlomov", + "Iftach Shoham", + "Alon Oved", + "Ido Levy", + "Sami Marreed", + "Harold Ship", + "Offer Akrabi", + "Sergey Zeltyn", + "Avi Yaeli", + "Nir Mashkif" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-safety", + "computer-use", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.20874", + "source": "arxiv", + "source_id": "arxiv:2605.20874", + "pdf_url": "https://arxiv.org/pdf/2605.20874", + "primary_query": "planning-agent" + }, + { + "id": "2605.20306", + "title": "WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents", + "url": "https://arxiv.org/abs/2605.20306", + "published": "2026-05-19", + "updated": "2026-06-02", + "authors": [ + "Bingnan Liu", + "Chenhang Cui", + "Rui Huang", + "Jiani Luo", + "Zhirong Shen", + "Tinghao Wang", + "Xiande Huang", + "Lingbei Meng", + "Fei Shen", + "An Zhang" + ], + "categories": [ + "cs.CV", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.20306", + "source": "arxiv", + "source_id": "arxiv:2605.20306", + "pdf_url": "https://arxiv.org/pdf/2605.20306", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.18672", + "title": "Position: A Three-Layer Probabilistic Assume-Guarantee Architecture Is Structurally Required for Safe LLM Agent Deployment", + "url": "https://arxiv.org/abs/2605.18672", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "S. Bensalem", + "Y. Dong", + "M. Franzle", + "X. Huang", + "J. Kroger", + "D. Nickovic", + "A. Nouri", + "R. Roy", + "C. Wu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.18672", + "source": "arxiv", + "source_id": "arxiv:2605.18672", + "pdf_url": "https://arxiv.org/pdf/2605.18672", + "primary_query": "agent-safety" + }, + { + "id": "2605.17348", + "title": "Taming \"Zombie'' Agents: A Markov State-Aware Framework for Resilient Multi-Agent Evolution", + "url": "https://arxiv.org/abs/2605.17348", + "published": "2026-05-17", + "updated": "2026-05-17", + "authors": [ + "Taolin Zhang", + "Pukun Zhao", + "Qizhou Chen", + "Jiuheng Wan", + "Chen Chen", + "Xiaofeng He", + "Chengyu Wang", + "Richang Hong" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-safety", + "memory", + "multi-agent", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.17348", + "source": "arxiv", + "source_id": "arxiv:2605.17348", + "pdf_url": "https://arxiv.org/pdf/2605.17348", + "primary_query": "agent-memory" + }, + { + "id": "2605.17453", + "title": "Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback", + "url": "https://arxiv.org/abs/2605.17453", + "published": "2026-05-17", + "updated": "2026-05-17", + "authors": [ + "Lecheng Yan", + "Ruizhe Li", + "Xicheng Han", + "Wenxi Li", + "Binwu Wang", + "Longyue Wang", + "Chenyang Lyu", + "Guanhua Chen" + ], + "categories": [ + "cs.CR", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.17453", + "source": "arxiv", + "source_id": "arxiv:2605.17453", + "pdf_url": "https://arxiv.org/pdf/2605.17453", + "primary_query": "agent-safety" + }, + { + "id": "2605.23986", + "title": "MemForest: An Efficient Agent Memory System with Hierarchical Temporal Indexing", + "url": "https://arxiv.org/abs/2605.23986", + "published": "2026-05-16", + "updated": "2026-05-16", + "authors": [ + "Han Chen", + "Zining Zhang", + "Wenqi Pei", + "Bingsheng He", + "Ming Wu", + "Jason Zeng", + "Michael Heinrich", + "Wei Wu", + "Hongbao Zhang" + ], + "categories": [ + "cs.DB", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.23986", + "source": "arxiv", + "source_id": "arxiv:2605.23986", + "pdf_url": "https://arxiv.org/pdf/2605.23986", + "primary_query": "agent-memory" + }, + { + "id": "2605.28850", + "title": "Representation Signatures and Risk-Feedback Alignment in LLM Trading Agents", + "url": "https://arxiv.org/abs/2605.28850", + "published": "2026-05-16", + "updated": "2026-05-30", + "authors": [ + "Weicheng Xue" + ], + "categories": [ + "cs.LG", + "q-fin.CP" + ], + "topics": [ + "agent-safety", + "coding-agent", + "memory", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.28850", + "source": "arxiv", + "source_id": "arxiv:2605.28850", + "pdf_url": "https://arxiv.org/pdf/2605.28850", + "primary_query": "planning-agent" + }, + { + "id": "2605.14460", + "title": "Exploiting LLM Agent Supply Chains via Payload-less Skills", + "url": "https://arxiv.org/abs/2605.14460", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Xinyu Liu", + "Yukai Zhao", + "Xing Hu", + "Xin Xia" + ], + "categories": [ + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14460", + "source": "arxiv", + "source_id": "arxiv:2605.14460", + "pdf_url": "https://arxiv.org/pdf/2605.14460", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.14126", + "title": "Reinforcement Learning for Tool-Calling Agents in Fast Healthcare Interoperability Resources (FHIR)", + "url": "https://arxiv.org/abs/2605.14126", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Marius S. Knorr", + "Robert Müller", + "Jan P. Bremer", + "Nils Schweingruber" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.14126", + "source": "arxiv", + "source_id": "arxiv:2605.14126", + "pdf_url": "https://arxiv.org/pdf/2605.14126", + "primary_query": "planning-agent" + }, + { + "id": "2605.11882", + "title": "On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment", + "url": "https://arxiv.org/abs/2605.11882", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Bo Yin", + "Qi Li", + "Xinchao Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.11882", + "source": "arxiv", + "source_id": "arxiv:2605.11882", + "pdf_url": "https://arxiv.org/pdf/2605.11882", + "primary_query": "agent-safety" + }, + { + "id": "2605.11534", + "title": "PRISM: : Planning and Reasoning with Intent in Simulated Embodied Environments", + "url": "https://arxiv.org/abs/2605.11534", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Yunn Kang Lim", + "Pengzhan Sun", + "Ziyi Bai", + "Xun Xu", + "Angela Yao", + "Xulei Yang", + "Shijie Li" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.11534", + "source": "arxiv", + "source_id": "arxiv:2605.11534", + "pdf_url": "https://arxiv.org/pdf/2605.11534", + "primary_query": "planning-agent" + }, + { + "id": "2605.11388", + "title": "Deep Reasoning in General Purpose Agents via Structured Meta-Cognition", + "url": "https://arxiv.org/abs/2605.11388", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Dean Light", + "Michael Theologitis", + "Kshitish Ghate", + "Shuyue Stella Li", + "Benjamin Newman", + "Chirag Shah", + "Aylin Caliskan", + "Pang Wei Koh", + "Dan Suciu", + "Yulia Tsvetkov" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.11388", + "source": "arxiv", + "source_id": "arxiv:2605.11388", + "pdf_url": "https://arxiv.org/pdf/2605.11388", + "primary_query": "planning-agent" + }, + { + "id": "2605.11039", + "title": "The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck", + "url": "https://arxiv.org/abs/2605.11039", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Linfeng Fan", + "Ziwei Li", + "Yuan Tian", + "Yichen Wang", + "Rongsheng Li", + "Xiong Wang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.11039", + "source": "arxiv", + "source_id": "arxiv:2605.11039", + "pdf_url": "https://arxiv.org/pdf/2605.11039", + "primary_query": "agent-safety" + }, + { + "id": "2605.10763", + "title": "MATRA: Modeling the Attack Surface of Agentic AI Systems -- OpenClaw Case Study", + "url": "https://arxiv.org/abs/2605.10763", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Tim Van hamme", + "Thomas Vissers", + "Javier Carnerero-Cano", + "Mario Fritz", + "Emil C. Lupu", + "Lieven Desmet", + "Dinil Mon Divakaran" + ], + "categories": [ + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.10763", + "source": "arxiv", + "source_id": "arxiv:2605.10763", + "pdf_url": "https://arxiv.org/pdf/2605.10763", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.10365", + "title": "Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values", + "url": "https://arxiv.org/abs/2605.10365", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Haonan Dong", + "Qiguan Feng", + "Kehan Jiang", + "Haoran Ye", + "Xin Zhang", + "Guojie Song" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.10365", + "source": "arxiv", + "source_id": "arxiv:2605.10365", + "pdf_url": "https://arxiv.org/pdf/2605.10365", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.09168", + "title": "CIVeX: Causal Intervention Verification for Language Agents", + "url": "https://arxiv.org/abs/2605.09168", + "published": "2026-05-09", + "updated": "2026-05-09", + "authors": [ + "Fabio Rovai" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.09168", + "source": "arxiv", + "source_id": "arxiv:2605.09168", + "pdf_url": "https://arxiv.org/pdf/2605.09168", + "primary_query": "language-agent" + }, + { + "id": "2605.08763", + "title": "When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions", + "url": "https://arxiv.org/abs/2605.08763", + "published": "2026-05-09", + "updated": "2026-05-09", + "authors": [ + "Minfeng Qi", + "Tianqing Zhu", + "Zijie Xu", + "Congcong Zhu", + "Qin Wang", + "Wanlei Zhou" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.08763", + "source": "arxiv", + "source_id": "arxiv:2605.08763", + "pdf_url": "https://arxiv.org/pdf/2605.08763", + "primary_query": "planning-agent" + }, + { + "id": "2605.06078", + "title": "Milestone-Guided Policy Learning for Long-Horizon Language Agents", + "url": "https://arxiv.org/abs/2605.06078", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Zixuan Wang", + "Yuchen Yan", + "Hongxing Li", + "Teng Pan", + "Dingming Li", + "Ruiqing Zhang", + "Weiming Lu", + "Jun Xiao", + "Yueting Zhuang", + "Yongliang Shen" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.06078", + "source": "arxiv", + "source_id": "arxiv:2605.06078", + "pdf_url": "https://arxiv.org/pdf/2605.06078", + "primary_query": "language-agent" + }, + { + "id": "2605.03505", + "title": "LATS-RCA: Language Agent Tree Search for Root Cause Analysis in Microservices", + "url": "https://arxiv.org/abs/2605.03505", + "published": "2026-05-05", + "updated": "2026-06-12", + "authors": [ + "Alexander Naakka", + "Yuqing Wang", + "Mika V Mäntylä" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.03505", + "source": "arxiv", + "source_id": "arxiv:2605.03505", + "pdf_url": "https://arxiv.org/pdf/2605.03505", + "primary_query": "language-agent" + }, + { + "id": "2605.04107", + "title": "TSCG: Deterministic Tool-Schema Compilation for Agentic LLM Deployments", + "url": "https://arxiv.org/abs/2605.04107", + "published": "2026-05-04", + "updated": "2026-05-04", + "authors": [ + "Furkan Sakizli" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.04107", + "source": "arxiv", + "source_id": "arxiv:2605.04107", + "pdf_url": "https://arxiv.org/pdf/2605.04107", + "primary_query": "function-calling" + }, + { + "id": "2605.05242", + "title": "Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction", + "url": "https://arxiv.org/abs/2605.05242", + "published": "2026-05-03", + "updated": "2026-05-03", + "authors": [ + "Zhuofeng Li", + "Haoxiang Zhang", + "Cong Wei", + "Pan Lu", + "Ping Nie", + "Yi Lu", + "Yuyang Bai", + "Shangbin Feng", + "Hangxiao Zhu", + "Ming Zhong", + "Yuyu Zhang", + "Jianwen Xie", + "Yejin Choi", + "James Zou", + "Jiawei Han", + "Wenhu Chen", + "Jimmy Lin", + "Dongfu Jiang", + "Yu Zhang" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.05242", + "source": "arxiv", + "source_id": "arxiv:2605.05242", + "pdf_url": "https://arxiv.org/pdf/2605.05242", + "primary_query": "language-agent" + }, + { + "id": "2605.00081", + "title": "Alignment Contracts for Agentic Security Systems", + "url": "https://arxiv.org/abs/2605.00081", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Isaac David", + "Marco Guarnieri", + "Arthur Gervais" + ], + "categories": [ + "cs.CR", + "cs.LO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.00081", + "source": "arxiv", + "source_id": "arxiv:2605.00081", + "pdf_url": "https://arxiv.org/pdf/2605.00081", + "primary_query": "agent-safety" + }, + { + "id": "2604.27699", + "title": "Bridging Values and Behavior: A Hierarchical Framework for Proactive Embodied Agents", + "url": "https://arxiv.org/abs/2604.27699", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Chunhui Zhang", + "Yuxuan Wang", + "Aoyang Qin", + "Yi-Long Lu", + "Kunlun Wu", + "Yizhou Wang", + "Wei Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27699", + "source": "arxiv", + "source_id": "arxiv:2604.27699", + "pdf_url": "https://arxiv.org/pdf/2604.27699", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.26274", + "title": "Enforcing Benign Trajectories: A Behavioral Firewall for Structured-Workflow AI Agents", + "url": "https://arxiv.org/abs/2604.26274", + "published": "2026-04-29", + "updated": "2026-04-29", + "authors": [ + "Hung Dang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.26274", + "source": "arxiv", + "source_id": "arxiv:2604.26274", + "pdf_url": "https://arxiv.org/pdf/2604.26274", + "primary_query": "agent-safety" + }, + { + "id": "2604.24826", + "title": "A Comparative Evaluation of AI Agent Security Guardrails", + "url": "https://arxiv.org/abs/2604.24826", + "published": "2026-04-27", + "updated": "2026-04-27", + "authors": [ + "Qi Li", + "Jiu Li", + "Pingtao Wei", + "Jianjun Xu", + "Xueyi Wei", + "Jiwei Shi", + "Xuan Zhang", + "Yanhui Yang", + "Xiaodong Hui", + "Peng Xu", + "Lingquan Zhou" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.24826", + "source": "arxiv", + "source_id": "arxiv:2604.24826", + "pdf_url": "https://arxiv.org/pdf/2604.24826", + "primary_query": "agent-safety" + }, + { + "id": "2606.13686", + "title": "Benchmarking Web Agent Safety under E-commerce Deceptive Interfaces", + "url": "https://arxiv.org/abs/2606.13686", + "published": "2026-04-26", + "updated": "2026-04-26", + "authors": [ + "Zijing Shi", + "Meng Fang", + "Ling Chen" + ], + "categories": [ + "cs.CL", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.13686", + "source": "arxiv", + "source_id": "arxiv:2606.13686", + "pdf_url": "https://arxiv.org/pdf/2606.13686", + "primary_query": "agent-safety" + }, + { + "id": "2604.19821", + "title": "JTPRO: A Joint Tool-Prompt Reflective Optimization Framework for Language Agents", + "url": "https://arxiv.org/abs/2604.19821", + "published": "2026-04-20", + "updated": "2026-04-20", + "authors": [ + "Sandip Ghoshal", + "Anshul Mittal", + "Jyotika Singh", + "Miguel Ballesteros", + "Weiyi Sun", + "Fang Tu", + "Shailender Singh", + "Yassine Benajiba", + "Fahad Shah", + "Sujeeth Bharadwaj", + "Sujith Ravi", + "Dan Roth" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.19821", + "source": "arxiv", + "source_id": "arxiv:2604.19821", + "pdf_url": "https://arxiv.org/pdf/2604.19821", + "primary_query": "language-agent" + }, + { + "id": "2604.18718", + "title": "Towards Optimal Agentic Architectures for Offensive Security Tasks", + "url": "https://arxiv.org/abs/2604.18718", + "published": "2026-04-20", + "updated": "2026-04-20", + "authors": [ + "Isaac David", + "Arthur Gervais" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.18718", + "source": "arxiv", + "source_id": "arxiv:2604.18718", + "pdf_url": "https://arxiv.org/pdf/2604.18718", + "primary_query": "agent-safety" + }, + { + "id": "2604.12986", + "title": "Parallax: Why AI Agents That Think Must Never Act", + "url": "https://arxiv.org/abs/2604.12986", + "published": "2026-04-14", + "updated": "2026-04-14", + "authors": [ + "Joel Fokou" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.12986", + "source": "arxiv", + "source_id": "arxiv:2604.12986", + "pdf_url": "https://arxiv.org/pdf/2604.12986", + "primary_query": "agent-safety" + }, + { + "id": "2604.06972", + "title": "Differentiable Environment-Trajectory Co-Optimization for Safe Multi-Agent Navigation", + "url": "https://arxiv.org/abs/2604.06972", + "published": "2026-04-08", + "updated": "2026-04-08", + "authors": [ + "Zhan Gao", + "Gabriele Fadini", + "Stelian Coros", + "Amanda Prorok" + ], + "categories": [ + "cs.RO", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.06972", + "source": "arxiv", + "source_id": "arxiv:2604.06972", + "pdf_url": "https://arxiv.org/pdf/2604.06972", + "primary_query": "agent-safety" + }, + { + "id": "2604.04426", + "title": "ShieldNet: Network-Level Guardrails against Emerging Supply-Chain Injections in Agentic Systems", + "url": "https://arxiv.org/abs/2604.04426", + "published": "2026-04-06", + "updated": "2026-04-06", + "authors": [ + "Zhuowen Yuan", + "Zhaorun Chen", + "Zhen Xiang", + "Nathaniel D. Bastian", + "Seyyed Hadi Hashemi", + "Chaowei Xiao", + "Wenbo Guo", + "Bo Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.04426", + "source": "arxiv", + "source_id": "arxiv:2604.04426", + "pdf_url": "https://arxiv.org/pdf/2604.04426", + "primary_query": "agent-safety" + }, + { + "id": "2604.03098", + "title": "Co-Evolution of Policy and Internal Reward for Language Agents", + "url": "https://arxiv.org/abs/2604.03098", + "published": "2026-04-03", + "updated": "2026-04-03", + "authors": [ + "Xinyu Wang", + "Hanwei Wu", + "Jingwei Song", + "Shuyuan Zhang", + "Jiayi Zhang", + "Fanqi Kong", + "Tung Sum Thomas Kwok", + "Xiao-Wen Chang", + "Yuyu Luo", + "Chenglin Wu", + "Bang Liu" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.03098", + "source": "arxiv", + "source_id": "arxiv:2604.03098", + "pdf_url": "https://arxiv.org/pdf/2604.03098", + "primary_query": "language-agent" + }, + { + "id": "2603.15309", + "title": "CCTU: A Benchmark for Tool Use under Complex Constraints", + "url": "https://arxiv.org/abs/2603.15309", + "published": "2026-03-16", + "updated": "2026-03-16", + "authors": [ + "Junjie Ye", + "Guoqiang Zhang", + "Wenjie Fu", + "Tao Gui", + "Qi Zhang", + "Xuanjing Huang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.15309", + "source": "arxiv", + "source_id": "arxiv:2603.15309", + "pdf_url": "https://arxiv.org/pdf/2603.15309", + "primary_query": "function-calling" + }, + { + "id": "2603.11890", + "title": "QUARE: Quality-Aware Requirements Analysis through Multi-Agent Dialectical Negotiation", + "url": "https://arxiv.org/abs/2603.11890", + "published": "2026-03-12", + "updated": "2026-06-05", + "authors": [ + "Haowei Cheng", + "Milhan Kim", + "Foutse Khomh", + "Teeradaj Racharak", + "Nobukazu Yoshioka", + "Naoyasu Ubayashi", + "Hironori Washizaki" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.11890", + "source": "arxiv", + "source_id": "arxiv:2603.11890", + "pdf_url": "https://arxiv.org/pdf/2603.11890", + "primary_query": "agent-safety" + }, + { + "id": "2603.07557", + "title": "AgentRaft: Automated Detection of Data Over-Exposure in LLM Agents", + "url": "https://arxiv.org/abs/2603.07557", + "published": "2026-03-08", + "updated": "2026-03-08", + "authors": [ + "Yixi Lin", + "Jiangrong Wu", + "Yuhong Nan", + "Xueqiang Wang", + "Xinyuan Zhang", + "Zibin Zheng" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.07557", + "source": "arxiv", + "source_id": "arxiv:2603.07557", + "pdf_url": "https://arxiv.org/pdf/2603.07557", + "primary_query": "function-calling" + }, + { + "id": "2603.05578", + "title": "Tool-Genesis: A Task-Driven Tool Creation Benchmark for Self-Evolving Language Agent", + "url": "https://arxiv.org/abs/2603.05578", + "published": "2026-03-05", + "updated": "2026-03-05", + "authors": [ + "Bowei Xia", + "Mengkang Hu", + "Shijian Wang", + "Jiarui Jin", + "Wenxiang Jiao", + "Yuan Lu", + "Kexin Li", + "Ping Luo" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.05578", + "source": "arxiv", + "source_id": "arxiv:2603.05578", + "pdf_url": "https://arxiv.org/pdf/2603.05578", + "primary_query": "language-agent" + }, + { + "id": "2603.01712", + "title": "FT-Dojo: Towards Autonomous LLM Fine-Tuning with Language Agents", + "url": "https://arxiv.org/abs/2603.01712", + "published": "2026-03-02", + "updated": "2026-05-20", + "authors": [ + "Qizheng Li", + "Yifei Zhang", + "Xiao Yang", + "Xu Yang", + "Zhuo Wang", + "Weiqing Liu", + "Jiang Bian" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.01712", + "source": "arxiv", + "source_id": "arxiv:2603.01712", + "pdf_url": "https://arxiv.org/pdf/2603.01712", + "primary_query": "language-agent" + }, + { + "id": "2602.13379", + "title": "Unsafer in Many Turns: Benchmarking and Defending Multi-Turn Safety Risks in Tool-Using Agents", + "url": "https://arxiv.org/abs/2602.13379", + "published": "2026-02-13", + "updated": "2026-06-10", + "authors": [ + "Xu Li", + "Simon Yu", + "Minzhou Pan", + "Yiyou Sun", + "Bo Li", + "Dawn Song", + "Xue Lin", + "Weiyan Shi" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.13379", + "source": "arxiv", + "source_id": "arxiv:2602.13379", + "pdf_url": "https://arxiv.org/pdf/2602.13379", + "primary_query": "agent-safety" + }, + { + "id": "2602.11749", + "title": "AIR: Improving Agent Safety through Incident Response", + "url": "https://arxiv.org/abs/2602.11749", + "published": "2026-02-12", + "updated": "2026-06-20", + "authors": [ + "Zibo Xiao", + "Jun Sun", + "Junjie Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.11749", + "source": "arxiv", + "source_id": "arxiv:2602.11749", + "pdf_url": "https://arxiv.org/pdf/2602.11749", + "primary_query": "agent-safety" + }, + { + "id": "2602.18456", + "title": "Beyond single-channel agentic benchmarking", + "url": "https://arxiv.org/abs/2602.18456", + "published": "2026-02-05", + "updated": "2026-02-05", + "authors": [ + "Nelu D. Radpour" + ], + "categories": [ + "cs.CY", + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.18456", + "source": "arxiv", + "source_id": "arxiv:2602.18456", + "pdf_url": "https://arxiv.org/pdf/2602.18456", + "primary_query": "agent-safety" + }, + { + "id": "2601.14652", + "title": "MAS-Orchestra: Understanding and Improving Multi-Agent Reasoning Through Holistic Orchestration and Controlled Benchmarks", + "url": "https://arxiv.org/abs/2601.14652", + "published": "2026-01-21", + "updated": "2026-05-21", + "authors": [ + "Zixuan Ke", + "Yifei Ming", + "Austin Xu", + "Ryan Chin", + "Xuan-Phi Nguyen", + "Prathyusha Jwalapuram", + "Jiayu Wang", + "Semih Yavuz", + "Caiming Xiong", + "Shafiq Joty" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.14652", + "source": "arxiv", + "source_id": "arxiv:2601.14652", + "pdf_url": "https://arxiv.org/pdf/2601.14652", + "primary_query": "function-calling" + }, + { + "id": "2512.23647", + "title": "Nested Browser-Use Learning for Agentic Information Seeking", + "url": "https://arxiv.org/abs/2512.23647", + "published": "2025-12-29", + "updated": "2025-12-29", + "authors": [ + "Baixuan Li", + "Jialong Wu", + "Wenbiao Yin", + "Kuan Li", + "Zhongwang Zhang", + "Huifeng Yin", + "Zhengwei Tao", + "Liwen Zhang", + "Pengjun Xie", + "Jingren Zhou", + "Yong Jiang" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.23647", + "source": "arxiv", + "source_id": "arxiv:2512.23647", + "pdf_url": "https://arxiv.org/pdf/2512.23647", + "primary_query": "function-calling" + }, + { + "id": "2512.23611", + "title": "Close the Loop: Synthesizing Infinite Tool-Use Data via Multi-Agent Role-Playing", + "url": "https://arxiv.org/abs/2512.23611", + "published": "2025-12-29", + "updated": "2025-12-29", + "authors": [ + "Yuwen Li", + "Wei Zhang", + "Zelong Huang", + "Mason Yang", + "Jiajun Wu", + "Shawn Guo", + "Huahao Hu", + "Lingyi Sun", + "Jian Yang", + "Mingjie Tang", + "Byran Dai" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.23611", + "source": "arxiv", + "source_id": "arxiv:2512.23611", + "pdf_url": "https://arxiv.org/pdf/2512.23611", + "primary_query": "function-calling" + }, + { + "id": "2512.02605", + "title": "IACT: A Self-Organizing Recursive Model for General AI Agents: A Technical White Paper on the Architecture Behind kragent.ai", + "url": "https://arxiv.org/abs/2512.02605", + "published": "2025-12-02", + "updated": "2025-12-02", + "authors": [ + "Pengju Lu" + ], + "categories": [ + "cs.AI", + "cs.MA", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.02605", + "source": "arxiv", + "source_id": "arxiv:2512.02605", + "pdf_url": "https://arxiv.org/pdf/2512.02605", + "primary_query": "function-calling" + }, + { + "id": "2510.14548", + "title": "LLM Agents Beyond Utility: An Open-Ended Perspective", + "url": "https://arxiv.org/abs/2510.14548", + "published": "2025-10-16", + "updated": "2025-10-16", + "authors": [ + "Asen Nachkov", + "Xi Wang", + "Luc Van Gool" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.14548", + "source": "arxiv", + "source_id": "arxiv:2510.14548", + "pdf_url": "https://arxiv.org/pdf/2510.14548", + "primary_query": "function-calling" + }, + { + "id": "2509.26553", + "title": "Towards Reliable Benchmarking: A Contamination Free, Controllable Evaluation Framework for Multi-step LLM Function Calling", + "url": "https://arxiv.org/abs/2509.26553", + "published": "2025-09-30", + "updated": "2026-02-06", + "authors": [ + "Seiji Maekawa", + "Jackson Hassell", + "Pouya Pezeshkpour", + "Tom Mitchell", + "Estevam Hruschka" + ], + "categories": [ + "cs.CL", + "cs.PL" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.26553", + "source": "arxiv", + "source_id": "arxiv:2509.26553", + "pdf_url": "https://arxiv.org/pdf/2509.26553", + "primary_query": "function-calling" + }, + { + "id": "2509.14477", + "title": "Ticket-Bench: A Kickoff for Multilingual and Regionalized Agent Evaluation", + "url": "https://arxiv.org/abs/2509.14477", + "published": "2025-09-17", + "updated": "2025-09-17", + "authors": [ + "Thales Sales Almeida", + "João Guilherme Alves Santos", + "Thiago Laitz", + "Giovana Kerche Bonás" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.14477", + "source": "arxiv", + "source_id": "arxiv:2509.14477", + "pdf_url": "https://arxiv.org/pdf/2509.14477", + "primary_query": "function-calling" + }, + { + "id": "2509.02444", + "title": "AppCopilot: Toward General, Accurate, Long-Horizon, and Efficient Mobile Agent", + "url": "https://arxiv.org/abs/2509.02444", + "published": "2025-09-02", + "updated": "2025-10-17", + "authors": [ + "Jingru Fan", + "Yufan Dang", + "Jingyao Wu", + "Huatao Li", + "Runde Yang", + "Xiyuan Yang", + "Yuheng Wang", + "Chen Qian" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CV", + "cs.HC" + ], + "topics": [ + "computer-use", + "memory", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.02444", + "source": "arxiv", + "source_id": "arxiv:2509.02444", + "pdf_url": "https://arxiv.org/pdf/2509.02444", + "primary_query": "function-calling" + }, + { + "id": "2607.06140", + "title": "CurateEvo: Data-Curation Evolving for Agentic Post-Training", + "url": "https://arxiv.org/abs/2607.06140", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Dingzirui Wang", + "Xuanliang Zhang", + "Keyan Xu", + "Qingfu Zhu", + "Wanxiang Che" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "memory", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.06140", + "source": "arxiv", + "source_id": "arxiv:2607.06140", + "pdf_url": "https://arxiv.org/pdf/2607.06140", + "primary_query": "llm-agent" + }, + { + "id": "2607.06452", + "title": "From Voting to Agent Collaboration: Answer-Type-Aware LLM Pipelines for BioASQ 14b", + "url": "https://arxiv.org/abs/2607.06452", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Taeyun Roh", + "Eunha Lee", + "Wonjune Jang", + "Sohyun Chung", + "Junha Jung", + "Jaewoo Kang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.06452", + "source": "arxiv", + "source_id": "arxiv:2607.06452", + "pdf_url": "https://arxiv.org/pdf/2607.06452", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.05772", + "title": "Detecting Vulnerability-Inducing Commits via Multi-Stage Reasoning with LLM-Based Agents", + "url": "https://arxiv.org/abs/2607.05772", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Liyou Chen", + "Hailong Sun", + "Xiang Gao", + "Yue Pan" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.05772", + "source": "arxiv", + "source_id": "arxiv:2607.05772", + "pdf_url": "https://arxiv.org/pdf/2607.05772", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.05378", + "title": "CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents", + "url": "https://arxiv.org/abs/2607.05378", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Yujiang Li", + "Zhenyu Hou", + "Yi Jing", + "Jie Tang", + "Yuxiao Dong" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "coding-agent", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.05378", + "source": "arxiv", + "source_id": "arxiv:2607.05378", + "pdf_url": "https://arxiv.org/pdf/2607.05378", + "primary_query": "coding-agent" + }, + { + "id": "2607.05132", + "title": "When Agents Lie: Premeditation, Persistence, and Exploitation in Repeated Games", + "url": "https://arxiv.org/abs/2607.05132", + "published": "2026-07-06", + "updated": "2026-07-07", + "authors": [ + "Jerick Shi", + "Terry Jingcheng Zhang", + "Bernhard Schölkopf", + "Vincent Conitzer", + "Zhijing Jin" + ], + "categories": [ + "cs.CY", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.05132", + "source": "arxiv", + "source_id": "arxiv:2607.05132", + "pdf_url": "https://arxiv.org/pdf/2607.05132", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.04963", + "title": "STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training", + "url": "https://arxiv.org/abs/2607.04963", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Qiuyi Qi", + "Tian Liang", + "Mutian Bao", + "Jinjian Zhang", + "Dongnan Liu", + "Wei Zhou", + "Linjian Mo", + "Ming Kong", + "Jie Liu", + "Feng Zhang", + "Qiang Zhu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "planning", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04963", + "source": "arxiv", + "source_id": "arxiv:2607.04963", + "pdf_url": "https://arxiv.org/pdf/2607.04963", + "primary_query": "llm-agent" + }, + { + "id": "2607.05677", + "title": "From Conversation to Contribution: Characterizing Coding Agent in Open-Source Software", + "url": "https://arxiv.org/abs/2607.05677", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Zihan Fang", + "Yueke Zhang", + "Ningzhi Tang", + "Collin McMillan", + "Toby Jia-Jun Li", + "Yu Huang" + ], + "categories": [ + "cs.SE", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.05677", + "source": "arxiv", + "source_id": "arxiv:2607.05677", + "pdf_url": "https://arxiv.org/pdf/2607.05677", + "primary_query": "coding-agent" + }, + { + "id": "2607.05188", + "title": "Latent Programming Horizons in Coding Agents", + "url": "https://arxiv.org/abs/2607.05188", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "André Silva", + "Han Tu", + "Martin Monperrus" + ], + "categories": [ + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.05188", + "source": "arxiv", + "source_id": "arxiv:2607.05188", + "pdf_url": "https://arxiv.org/pdf/2607.05188", + "primary_query": "coding-agent" + }, + { + "id": "2607.04623", + "title": "Can LLMs Really Recover Microservice Failures? A Recovery-Aware Evaluation of Diagnosis-to-Action Reasoning", + "url": "https://arxiv.org/abs/2607.04623", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Jiaxing Qi", + "Zhongzhi Luan", + "Hongyu Zhang", + "Shaohan Huang", + "Carol Fung", + "Yongxin Tong", + "Hailong Yang", + "Depei Qian" + ], + "categories": [ + "cs.SE", + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.04623", + "source": "arxiv", + "source_id": "arxiv:2607.04623", + "pdf_url": "https://arxiv.org/pdf/2607.04623", + "primary_query": "rag-agent" + }, + { + "id": "2607.04470", + "title": "Regime-Conditional Stabilisation of LLM-Augmented Cooperative Multi-Agent Reinforcement Learning", + "url": "https://arxiv.org/abs/2607.04470", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Faid Keddouri", + "Sohaib Houhou", + "Aissa Boulmerka", + "Nadir Farhi" + ], + "categories": [ + "cs.LG", + "cs.AI", + "math.OC" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.04470", + "source": "arxiv", + "source_id": "arxiv:2607.04470", + "pdf_url": "https://arxiv.org/pdf/2607.04470", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.03853", + "title": "CogRad: A Cognitively-Inspired Multi-Agent Framework for Radiology Report Generation", + "url": "https://arxiv.org/abs/2607.03853", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Saif Ur Rehman Khan", + "Hasaan Maqsood", + "Sebastian Vollmer", + "Andreas Dengel", + "Muhammad Nabeel Asim" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.03853", + "source": "arxiv", + "source_id": "arxiv:2607.03853", + "pdf_url": "https://arxiv.org/pdf/2607.03853", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.02879", + "title": "MedCalc-Pro: Solving Complex Medical Calculations with LLM Agents", + "url": "https://arxiv.org/abs/2607.02879", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Siran Zhao", + "Ruihui Hou", + "Ziyue Huai", + "Chennuo Zhang", + "Tong Ruan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.02879", + "source": "arxiv", + "source_id": "arxiv:2607.02879", + "pdf_url": "https://arxiv.org/pdf/2607.02879", + "primary_query": "llm-agent" + }, + { + "id": "2607.03423", + "title": "Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies", + "url": "https://arxiv.org/abs/2607.03423", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Chris Schneider", + "Kriti Faujdar", + "Philipp Schoenegger", + "Ben Bariach" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2607.03423", + "source": "arxiv", + "source_id": "arxiv:2607.03423", + "pdf_url": "https://arxiv.org/pdf/2607.03423", + "primary_query": "ai-agent" + }, + { + "id": "2607.03162", + "title": "APeB: Benchmarking Personalization Ability of Large Language Model Agents", + "url": "https://arxiv.org/abs/2607.03162", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Garry Yang", + "Zizhe Chen", + "Xinru Chen", + "Yongqiang Chen", + "Jianxiang Wang", + "Deyu Zou", + "Linyi Ding", + "Jialiang Wu", + "Yunzhong He", + "Yu Gong", + "James Cheng", + "Huaixiao Tou" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.03162", + "source": "arxiv", + "source_id": "arxiv:2607.03162", + "pdf_url": "https://arxiv.org/pdf/2607.03162", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02882", + "title": "Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference", + "url": "https://arxiv.org/abs/2607.02882", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Xuyan Ma", + "Yawen Wang", + "Junjie Wang", + "Xiaofei Xie", + "Boyu Wu", + "Mingyang Li", + "Dandan Wang", + "Qing Wang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02882", + "source": "arxiv", + "source_id": "arxiv:2607.02882", + "pdf_url": "https://arxiv.org/pdf/2607.02882", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03628", + "title": "Swarm-Driven Multi-Agent Reasoning for Smart City Security", + "url": "https://arxiv.org/abs/2607.03628", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Saeid Jamshidi", + "Kawser Wazed Nafi", + "Carol Fung", + "Foutse Khomh" + ], + "categories": [ + "cs.CR", + "cs.MA" + ], + "topics": [ + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.03628", + "source": "arxiv", + "source_id": "arxiv:2607.03628", + "pdf_url": "https://arxiv.org/pdf/2607.03628", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.01600", + "title": "BOUNDARY_SYNC: Measuring Communication-Induced Representational Coupling in Multi-Agent LLM Systems", + "url": "https://arxiv.org/abs/2607.01600", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Zewen Liu" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "multi-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.01600", + "source": "arxiv", + "source_id": "arxiv:2607.01600", + "pdf_url": "https://arxiv.org/pdf/2607.01600", + "primary_query": "llm-agent" + }, + { + "id": "2607.02210", + "title": "Criticality-Based Guard Rail Validation for AI Agent Decisions in Autonomous Telecom Networks", + "url": "https://arxiv.org/abs/2607.02210", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Ravi Kant Sharma" + ], + "categories": [ + "cs.AI", + "cs.NI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.02210", + "source": "arxiv", + "source_id": "arxiv:2607.02210", + "pdf_url": "https://arxiv.org/pdf/2607.02210", + "primary_query": "ai-agent" + }, + { + "id": "2607.01661", + "title": "Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry", + "url": "https://arxiv.org/abs/2607.01661", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Yuante Li", + "Yicheng Tao", + "Kate Zhang", + "Taozhi Wang", + "Gefei Gu", + "Yaxin Zhou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.01661", + "source": "arxiv", + "source_id": "arxiv:2607.01661", + "pdf_url": "https://arxiv.org/pdf/2607.01661", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00692", + "title": "Self-GC: Self-Governing Context for Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2607.00692", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Xubin Hao", + "Hongjin Meng", + "Xin Yin", + "Jiawei Zhu", + "Chenpeng Cao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "planning", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.00692", + "source": "arxiv", + "source_id": "arxiv:2607.00692", + "pdf_url": "https://arxiv.org/pdf/2607.00692", + "primary_query": "llm-agent" + }, + { + "id": "2607.00297", + "title": "EPC: A Standardized Protocol for Measuring Evaluator Preference Dynamics in LLM Agent Systems", + "url": "https://arxiv.org/abs/2607.00297", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Zewen Liu" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00297", + "source": "arxiv", + "source_id": "arxiv:2607.00297", + "pdf_url": "https://arxiv.org/pdf/2607.00297", + "primary_query": "llm-agent" + }, + { + "id": "2607.01523", + "title": "Multi-Head Recurrent Memory Agents", + "url": "https://arxiv.org/abs/2607.01523", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Jiatong Li", + "Samuel Yeh", + "Sharon Li" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2607.01523", + "source": "arxiv", + "source_id": "arxiv:2607.01523", + "pdf_url": "https://arxiv.org/pdf/2607.01523", + "primary_query": "agent-memory" + }, + { + "id": "2607.01211", + "title": "Are Performance-Optimization Benchmarks Reliably Measuring Coding Agents?", + "url": "https://arxiv.org/abs/2607.01211", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Zhi Chen", + "Zhensu Sun", + "Yuling Shi", + "David Lo", + "Lingxiao Jiang" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.01211", + "source": "arxiv", + "source_id": "arxiv:2607.01211", + "pdf_url": "https://arxiv.org/pdf/2607.01211", + "primary_query": "coding-agent" + }, + { + "id": "2607.00918", + "title": "From Personas to Plot: Character-Grounded Multi-Agent Story Generation for Long-Form Narratives", + "url": "https://arxiv.org/abs/2607.00918", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Aayush Aluru", + "Chloe Ho", + "Muhammad Hammouri", + "Kerry Luo", + "Myra Malik", + "Ryan Lagasse", + "Arjun Bahuguna", + "Vasu Sharma" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00918", + "source": "arxiv", + "source_id": "arxiv:2607.00918", + "pdf_url": "https://arxiv.org/pdf/2607.00918", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00604", + "title": "Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives", + "url": "https://arxiv.org/abs/2607.00604", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Xianchao Xiu", + "Chong Shen", + "Yanjiao Zhu", + "Wanquan Liu" + ], + "categories": [ + "math.OC" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00604", + "source": "arxiv", + "source_id": "arxiv:2607.00604", + "pdf_url": "https://arxiv.org/pdf/2607.00604", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00972", + "title": "Bayesian Uncertainty Propagation for Agentic RAG Pipelines: A Proof-of-Concept Study on Multi-Hop Question Answering", + "url": "https://arxiv.org/abs/2607.00972", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Louis Donaldson", + "Connor Walker", + "Koorosh Aslansefat", + "Yiannis Papadopoulos" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.00972", + "source": "arxiv", + "source_id": "arxiv:2607.00972", + "pdf_url": "https://arxiv.org/pdf/2607.00972", + "primary_query": "rag-agent" + }, + { + "id": "2607.00422", + "title": "KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems", + "url": "https://arxiv.org/abs/2607.00422", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Chanwoo Choi", + "Euntae Kim", + "Kyuho Lee", + "Youngsam Chun", + "Jinhee Jeong", + "Eunmi Kim", + "Myunggyo Oh", + "Junseo Jang", + "Buru Chang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.00422", + "source": "arxiv", + "source_id": "arxiv:2607.00422", + "pdf_url": "https://arxiv.org/pdf/2607.00422", + "primary_query": "rag-agent" + }, + { + "id": "2606.31635", + "title": "A Tutorial on Autonomous Fault-Tolerant Control Using Knowledge-Grounded LLM Agents", + "url": "https://arxiv.org/abs/2606.31635", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Javal Vyas", + "Milapji Singh Gill", + "Artan Markaj", + "Felix Gehlhoff", + "Mehmet Mercangöz" + ], + "categories": [ + "eess.SY", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-safety", + "planning", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.31635", + "source": "arxiv", + "source_id": "arxiv:2606.31635", + "pdf_url": "https://arxiv.org/pdf/2606.31635", + "primary_query": "llm-agent" + }, + { + "id": "2606.31227", + "title": "Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming", + "url": "https://arxiv.org/abs/2606.31227", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yong Yang", + "Xing Zheng", + "Huiyu Wu", + "Huangsheng Cheng", + "Xiaorong Shi", + "Jing Guo", + "Bo Yang", + "Yi Zhou", + "Xiangfan Wu", + "Zonghao Ying" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "ai-agent" + ], + "arxiv_id": "2606.31227", + "source": "arxiv", + "source_id": "arxiv:2606.31227", + "pdf_url": "https://arxiv.org/pdf/2606.31227", + "primary_query": "agent-safety" + }, + { + "id": "2607.00255", + "title": "SLM, LLM or Agentic AI? Toward Intelligent UAV-Enabled WPT Systems in Low-Altitude Economy Networks", + "url": "https://arxiv.org/abs/2607.00255", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Feibo Jiang", + "Li Dong", + "Lei Mao", + "Kezhi Wang", + "Xianbin Wang", + "Abbas Jamalipour" + ], + "categories": [ + "cs.IT" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.00255", + "source": "arxiv", + "source_id": "arxiv:2607.00255", + "pdf_url": "https://arxiv.org/pdf/2607.00255", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31648", + "title": "Think in English, Answer in Korean: Efficient Adaptation of Multilingual Tool-Using Agents", + "url": "https://arxiv.org/abs/2606.31648", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Utsav Garg", + "Sungjin Hong", + "Jason Jung", + "Justin Lee", + "Shaan Desai", + "Joon Hee Kim", + "Anirudh Shrinivason", + "Edmond Wen", + "Susie Park" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "memory", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "function-calling", + "tool-use" + ], + "arxiv_id": "2606.31648", + "source": "arxiv", + "source_id": "arxiv:2606.31648", + "pdf_url": "https://arxiv.org/pdf/2606.31648", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02577", + "title": "Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation", + "url": "https://arxiv.org/abs/2607.02577", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Vishvesh Bhat", + "Jay Vaghasiya", + "Muhammad Ahmed Mohsin", + "Asad Aali" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.02577", + "source": "arxiv", + "source_id": "arxiv:2607.02577", + "pdf_url": "https://arxiv.org/pdf/2607.02577", + "primary_query": "tool-use" + }, + { + "id": "2606.31314", + "title": "A Novel Method for Differential-Algebraic Dynamic Model Discovery in Power Systems: An LLM-Based Multi-Agent Collaborative Framework", + "url": "https://arxiv.org/abs/2606.31314", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Xinming Wang", + "Fan Tang", + "Yingli Wei", + "Yakun He", + "Zhe Liu", + "Ping Jiang", + "Haoyu Wu", + "Zihan Guo", + "Chao Shen" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31314", + "source": "arxiv", + "source_id": "arxiv:2606.31314", + "pdf_url": "https://arxiv.org/pdf/2606.31314", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.30840", + "title": "Contrastive Reflection for Iterative Prompt Optimization", + "url": "https://arxiv.org/abs/2606.30840", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Derek Koh", + "Jinghui Mo", + "Benjamin H. Le", + "Jiening Zhan", + "Baofen Zheng", + "Kevin Bevis", + "Nathaniel C. Owen", + "Lauren Elizabeth Charney", + "Wenqiong Liu", + "Jingwei Wu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent" + ], + "arxiv_id": "2606.30840", + "source": "arxiv", + "source_id": "arxiv:2606.30840", + "pdf_url": "https://arxiv.org/pdf/2606.30840", + "primary_query": "ai-agent" + }, + { + "id": "2606.30454", + "title": "Collective cooperation without individual fidelity in LLM agents", + "url": "https://arxiv.org/abs/2606.30454", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Henrique Ferraz de Arruda", + "Carlos Gracia Lázaro", + "Alberto Aleta", + "Yamir Moreno" + ], + "categories": [ + "physics.soc-ph", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.30454", + "source": "arxiv", + "source_id": "arxiv:2606.30454", + "pdf_url": "https://arxiv.org/pdf/2606.30454", + "primary_query": "llm-agent" + }, + { + "id": "2606.30005", + "title": "LLM Agents Are Latent Context Managers: Eliciting Self-Managed Context via a Proprioceptive Dashboard", + "url": "https://arxiv.org/abs/2606.30005", + "published": "2026-06-29", + "updated": "2026-07-05", + "authors": [ + "Binyan Xu", + "Haitao Li", + "Kehuan Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "memory", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.30005", + "source": "arxiv", + "source_id": "arxiv:2606.30005", + "pdf_url": "https://arxiv.org/pdf/2606.30005", + "primary_query": "llm-agent" + }, + { + "id": "2606.29762", + "title": "Do Recommendation Algorithms Work When Users Are LLM Agents? A Case Study on Moltbook", + "url": "https://arxiv.org/abs/2606.29762", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Daming Li", + "Simeng Han", + "Jialu Zhang" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent" + ], + "arxiv_id": "2606.29762", + "source": "arxiv", + "source_id": "arxiv:2606.29762", + "pdf_url": "https://arxiv.org/pdf/2606.29762", + "primary_query": "ai-agent" + }, + { + "id": "2606.30266", + "title": "Towards Continual Motion-Language Agents: LoRA Variants for Incremental Motion Understanding and Generation", + "url": "https://arxiv.org/abs/2606.30266", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Bertram Taetz", + "Hugo Albuquerque Cosme da Silva", + "Gabriele Bleser-Taetz" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "language-agent" + ], + "arxiv_id": "2606.30266", + "source": "arxiv", + "source_id": "arxiv:2606.30266", + "pdf_url": "https://arxiv.org/pdf/2606.30266", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.30185", + "title": "Dynamo: Dynamic Skill-Tool Evolution for Vision-Language Agents", + "url": "https://arxiv.org/abs/2606.30185", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yutao Sun", + "Yanting Miao", + "Hao-Xuan Ma", + "Mengyu Zhou", + "Mingshuai Chen", + "Tiancheng Zhao", + "Dexin Wang", + "Lei Lv", + "Li Xu", + "Xiaoxi Jiang", + "Guanjun Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.30185", + "source": "arxiv", + "source_id": "arxiv:2606.30185", + "pdf_url": "https://arxiv.org/pdf/2606.30185", + "primary_query": "language-agent" + }, + { + "id": "2606.30755", + "title": "Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens", + "url": "https://arxiv.org/abs/2606.30755", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Peizhi Niu", + "Wenjie Qu", + "Shangding Gu", + "Tianneng Shi", + "Yuankai Li", + "Ahmad Tawaha", + "Hend Alzahrani", + "Vincent Siu", + "Boyi Li", + "Chenguang Wang", + "Jiaheng Zhang", + "Basel Alomair", + "Ming Jin", + "Muhao Chen", + "Chi Wang", + "Costas Spanos", + "Dawn Song" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "ai-agent" + ], + "arxiv_id": "2606.30755", + "source": "arxiv", + "source_id": "arxiv:2606.30755", + "pdf_url": "https://arxiv.org/pdf/2606.30755", + "primary_query": "agent-safety" + }, + { + "id": "2606.29894", + "title": "SABER-Math: Automated Benchmark for Information Retrieval Evaluation in Mathematics", + "url": "https://arxiv.org/abs/2606.29894", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Nikolay Georgiev", + "Maria Drencheva", + "Kseniia Ibragimova", + "Ivo Petrov", + "Dimitar I. Dimitrov", + "Martin Vechev" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.29894", + "source": "arxiv", + "source_id": "arxiv:2606.29894", + "pdf_url": "https://arxiv.org/pdf/2606.29894", + "primary_query": "agentic-ai" + }, + { + "id": "2606.29961", + "title": "DuoMem: Towards Capable On-Device Memory Agents via Dual-Space Distillation", + "url": "https://arxiv.org/abs/2606.29961", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Peyman Hosseini", + "Ondrej Bohdal", + "Ahmed Alajrami", + "Andrea Maracani", + "Ignacio Castro", + "Matthew Purver", + "Mete Ozay", + "Savas Ozkan", + "Taha Ceritli" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.29961", + "source": "arxiv", + "source_id": "arxiv:2606.29961", + "pdf_url": "https://arxiv.org/pdf/2606.29961", + "primary_query": "agent-memory" + }, + { + "id": "2606.30119", + "title": "On the Internet, Nobody Knows You're an LLM Bot: Unmasking Web Agents with Multi-Layer Fingerprinting", + "url": "https://arxiv.org/abs/2606.30119", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Iliana Fayolle", + "Sihem Bouhenniche", + "Samuel Pélissier", + "Pierre Laperdrix", + "Clémentine Maurice", + "Walter Rudametkin" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.30119", + "source": "arxiv", + "source_id": "arxiv:2606.30119", + "pdf_url": "https://arxiv.org/pdf/2606.30119", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.30602", + "title": "MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.30602", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Kunyang Li", + "Kyle Domico", + "Jonathan Gregory", + "Patrick McDaniel" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30602", + "source": "arxiv", + "source_id": "arxiv:2606.30602", + "pdf_url": "https://arxiv.org/pdf/2606.30602", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29932", + "title": "SAGA: Scene-Aware, Goal-Evolving Agents for Long-Horizon CivRealm Strategy Planning", + "url": "https://arxiv.org/abs/2606.29932", + "published": "2026-06-29", + "updated": "2026-07-02", + "authors": [ + "Tianyu Jin", + "Shuo Chen", + "Yida Wang", + "Liuyu Xiang", + "Yingzhuo Liu", + "Zhiyao Jiang", + "Yexin Li", + "Zhaofeng He" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.29932", + "source": "arxiv", + "source_id": "arxiv:2606.29932", + "pdf_url": "https://arxiv.org/pdf/2606.29932", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29746", + "title": "DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification", + "url": "https://arxiv.org/abs/2606.29746", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Maolin Liu", + "Fanyu Xu", + "Ruoqing Xu", + "Jiahang Zhang", + "Hao Wang", + "Rui Wang" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.29746", + "source": "arxiv", + "source_id": "arxiv:2606.29746", + "pdf_url": "https://arxiv.org/pdf/2606.29746", + "primary_query": "rag-agent" + }, + { + "id": "2606.29225", + "title": "PolicyGuard: A Dialogue-Grounded Sub-Agent Verifier for Policy Adherence in LLM Agents", + "url": "https://arxiv.org/abs/2606.29225", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Seongjae Kang", + "Taehyung Yu", + "Sung Ju Hwang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29225", + "source": "arxiv", + "source_id": "arxiv:2606.29225", + "pdf_url": "https://arxiv.org/pdf/2606.29225", + "primary_query": "llm-agent" + }, + { + "id": "2606.29142", + "title": "Agent Security Meets Regulatory Reality -- A Practitioner Systematization of Autonomous-Agent Threats and Controls in Regulated Financial Systems", + "url": "https://arxiv.org/abs/2606.29142", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Krishna Mohan", + "Guda Nagavenkata Srinivasa" + ], + "categories": [ + "cs.CY", + "cs.SE" + ], + "topics": [ + "agent-safety", + "computer-use", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "autonomous-agent-llm", + "rag-agent" + ], + "arxiv_id": "2606.29142", + "source": "arxiv", + "source_id": "arxiv:2606.29142", + "pdf_url": "https://arxiv.org/pdf/2606.29142", + "primary_query": "agent-safety" + }, + { + "id": "2606.28733", + "title": "Agentic Abstention: Do Agents Know When to Stop Instead of Act?", + "url": "https://arxiv.org/abs/2606.28733", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Han Luo", + "Bingbing Wen", + "Lucy Lu Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.28733", + "source": "arxiv", + "source_id": "arxiv:2606.28733", + "pdf_url": "https://arxiv.org/pdf/2606.28733", + "primary_query": "llm-agent" + }, + { + "id": "2606.28679", + "title": "Capability Gates Are Not Authorization: Confused-Deputy Failures in LLM Agent Frameworks", + "url": "https://arxiv.org/abs/2606.28679", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "David Mellafe Zuvic" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.28679", + "source": "arxiv", + "source_id": "arxiv:2606.28679", + "pdf_url": "https://arxiv.org/pdf/2606.28679", + "primary_query": "llm-agent" + }, + { + "id": "2606.29026", + "title": "Preventing Error Propagation in Multi-Agent AI through Runtime Monitoring", + "url": "https://arxiv.org/abs/2606.29026", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Shahnewaz Karim Sakib", + "Anindya Bijoy Das" + ], + "categories": [ + "cs.AI", + "cs.ET" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.29026", + "source": "arxiv", + "source_id": "arxiv:2606.29026", + "pdf_url": "https://arxiv.org/pdf/2606.29026", + "primary_query": "agentic-ai" + }, + { + "id": "2606.28739", + "title": "Agent Safety Is Action Alignment", + "url": "https://arxiv.org/abs/2606.28739", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Shawn Li", + "Yue Zhao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.28739", + "source": "arxiv", + "source_id": "arxiv:2606.28739", + "pdf_url": "https://arxiv.org/pdf/2606.28739", + "primary_query": "agent-safety" + }, + { + "id": "2606.27632", + "title": "Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety", + "url": "https://arxiv.org/abs/2606.27632", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Ting Ma", + "Xiufeng Huang", + "Benlei Cui", + "Xiaowen Xu", + "Shikai Qiu", + "Ruijie Jian", + "Hongxing Li", + "Guanghui Wang", + "Longtao Huang", + "Haiwen Hong", + "Haolei Xu", + "Wenjing Jiang", + "Ziwen Xu", + "Zhaoyu Fan", + "Shaoxuan He", + "Chuxi Xiao", + "Yujian Li", + "Xinyue Chen", + "Chunyang Chai", + "Wenxuan Liu", + "Ziheng Wang", + "Dongjie Zhang", + "Yangfan Zhou", + "Libin Dong", + "Yupeng Cao", + "Xiaoqian Xia", + "Jing Wang", + "Zhe Jiang", + "Zhenan Ye", + "Guang Yang", + "Bin Liu", + "Wei Peng", + "Ziqiang Zhu", + "Meihui Lian", + "Kaiwen Lv Kacuila", + "Haidong Ding", + "Bingyu Zhu", + "Yan Wang", + "Hai Zhao", + "Xuan Jin", + "Wei Zhao", + "Pengfei Sun", + "Wei Wang", + "Huiming Zhang", + "Bin Li", + "Hui Xue" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.27632", + "source": "arxiv", + "source_id": "arxiv:2606.27632", + "pdf_url": "https://arxiv.org/pdf/2606.27632", + "primary_query": "tool-use" + }, + { + "id": "2606.26806", + "title": "Memory Depth, Not Memory Access: Selective Parametric Consolidation for Long-Running Language Agents", + "url": "https://arxiv.org/abs/2606.26806", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Haoliang Han" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.26806", + "source": "arxiv", + "source_id": "arxiv:2606.26806", + "pdf_url": "https://arxiv.org/pdf/2606.26806", + "primary_query": "language-agent" + }, + { + "id": "2606.27154", + "title": "OpenRCA 2.0: From Outcome Labels to Causal Process Supervision", + "url": "https://arxiv.org/abs/2606.27154", + "published": "2026-06-25", + "updated": "2026-06-30", + "authors": [ + "Aoyang Fang", + "Yifan Yang", + "Jin'ao Shang", + "Qisheng Lu", + "Junjielung Xu", + "Rui Wang", + "Songhan Zhang", + "Yuzhong Zhang", + "Boxi Yu", + "Pinjia He" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.27154", + "source": "arxiv", + "source_id": "arxiv:2606.27154", + "pdf_url": "https://arxiv.org/pdf/2606.27154", + "primary_query": "tool-use" + }, + { + "id": "2606.27492", + "title": "QueenBee Planner: Skill-Evolving Communication Topologies for Token-Efficient LLM Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.27492", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Congjia Tian", + "Yuhang Yao", + "Jiaming Cui" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.27492", + "source": "arxiv", + "source_id": "arxiv:2606.27492", + "pdf_url": "https://arxiv.org/pdf/2606.27492", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.26758", + "title": "EGG: An Expert-Guided Agent Framework for Kernel Generation", + "url": "https://arxiv.org/abs/2606.26758", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Yaochen Han", + "Ke Fan", + "Hongxu Jiang", + "Wanqi Xu", + "Weiyu Xie", + "Runhua Zhang", + "Chenhui Zhu", + "Yixiang Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "memory", + "multi-agent", + "rag", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.26758", + "source": "arxiv", + "source_id": "arxiv:2606.26758", + "pdf_url": "https://arxiv.org/pdf/2606.26758", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.26205", + "title": "Knowledge-augmented Agentic AI for Mental Health Medication Information Seeking", + "url": "https://arxiv.org/abs/2606.26205", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Huizi Yu", + "Jian Liu", + "Wenkong Wang", + "Lingyao Li", + "Jiayan Zhou", + "Zhaoqian Xue", + "Xiang Li", + "Xinxin Lin", + "Zhiying Liang", + "Zhuoru Wu", + "Siyuan Ma", + "Xin Ma", + "Lizhou Fan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.26205", + "source": "arxiv", + "source_id": "arxiv:2606.26205", + "pdf_url": "https://arxiv.org/pdf/2606.26205", + "primary_query": "agentic-ai" + }, + { + "id": "2606.25899", + "title": "Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs", + "url": "https://arxiv.org/abs/2606.25899", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Adeeb Zaman", + "Erik Nordby", + "Fred Heiding" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.25899", + "source": "arxiv", + "source_id": "arxiv:2606.25899", + "pdf_url": "https://arxiv.org/pdf/2606.25899", + "primary_query": "agentic-ai" + }, + { + "id": "2606.25484", + "title": "From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities", + "url": "https://arxiv.org/abs/2606.25484", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Meng Jin", + "Melanie Handrich", + "Simone Martinenz", + "Nicholas Hoeser", + "Ziyue Li" + ], + "categories": [ + "cs.CY", + "econ.GN", + "stat.AP" + ], + "topics": [ + "coding-agent", + "computer-use", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.25484", + "source": "arxiv", + "source_id": "arxiv:2606.25484", + "pdf_url": "https://arxiv.org/pdf/2606.25484", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26300", + "title": "The Verification Horizon: No Silver Bullet for Coding Agent Rewards", + "url": "https://arxiv.org/abs/2606.26300", + "published": "2026-06-24", + "updated": "2026-06-29", + "authors": [ + "Binghai Wang", + "Chenlong Zhang", + "Dayiheng Liu", + "Jiajun Zhang", + "Jiawei Chen", + "Mingze Li", + "Mouxiang Chen", + "Rongyao Fang", + "Siyuan Zhang", + "Xuwu Wang", + "Yuheng Jing", + "Zeyao Ma", + "Zeyu Cui" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.26300", + "source": "arxiv", + "source_id": "arxiv:2606.26300", + "pdf_url": "https://arxiv.org/pdf/2606.26300", + "primary_query": "coding-agent" + }, + { + "id": "2606.25705", + "title": "GUI agent: Guided Exploration of User-Sensitive Screens", + "url": "https://arxiv.org/abs/2606.25705", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Aradhana Nayak", + "Mussadiq Nazeer", + "Wang Peng", + "Feng Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.25705", + "source": "arxiv", + "source_id": "arxiv:2606.25705", + "pdf_url": "https://arxiv.org/pdf/2606.25705", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.25656", + "title": "Is GraphRAG Needed? From Basic RAG to Graph-/Agentic Solutions with Context Optimization", + "url": "https://arxiv.org/abs/2606.25656", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Long Chen", + "Ryan Razkenari", + "Yuxuan Zhou", + "Yuan Tian", + "Rahul Ghosh", + "Venkatesh Pappakrishnan", + "Disha Ahuja", + "Vidya Sagar Ravipati" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.25656", + "source": "arxiv", + "source_id": "arxiv:2606.25656", + "pdf_url": "https://arxiv.org/pdf/2606.25656", + "primary_query": "rag-agent" + }, + { + "id": "2606.25189", + "title": "ActPlane: Programmable OS-Level Policy Enforcement for Agent Harnesses", + "url": "https://arxiv.org/abs/2606.25189", + "published": "2026-06-23", + "updated": "2026-06-30", + "authors": [ + "Yusheng Zheng", + "Tianyuan Wu", + "Quanzhi Fu", + "Tong Yu", + "Wenan Mao", + "Tao Ma", + "Dan Williams", + "Wei Wang", + "Andi Quinn" + ], + "categories": [ + "cs.OS" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.25189", + "source": "arxiv", + "source_id": "arxiv:2606.25189", + "pdf_url": "https://arxiv.org/pdf/2606.25189", + "primary_query": "ai-agent" + }, + { + "id": "2606.24402", + "title": "Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents", + "url": "https://arxiv.org/abs/2606.24402", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Juho Park", + "Hyunmin Choi", + "Kevin Nam" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2606.24402", + "source": "arxiv", + "source_id": "arxiv:2606.24402", + "pdf_url": "https://arxiv.org/pdf/2606.24402", + "primary_query": "ai-agent" + }, + { + "id": "2606.24235", + "title": "SP-Mind: An Autonomous Reasoning Agent for Spatial Proteomics Analysis", + "url": "https://arxiv.org/abs/2606.24235", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yucheng Yuan", + "Yuanfeng Ji", + "Zhongxiao Li", + "Ruijiang Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.24235", + "source": "arxiv", + "source_id": "arxiv:2606.24235", + "pdf_url": "https://arxiv.org/pdf/2606.24235", + "primary_query": "ai-agent" + }, + { + "id": "2606.25206", + "title": "RAVEN: Long-Horizon Reasoning & Navigation with a Visuo-Spatio-Temporal Memory", + "url": "https://arxiv.org/abs/2606.25206", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yixun Hu", + "Zhicheng Zheng", + "Lihan Zha", + "Chunwei Xing", + "Rajdeep Singh", + "Omar Hossain", + "Antonio Loquercio", + "Dhruv Shah" + ], + "categories": [ + "cs.RO", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.25206", + "source": "arxiv", + "source_id": "arxiv:2606.25206", + "pdf_url": "https://arxiv.org/pdf/2606.25206", + "primary_query": "agent-memory" + }, + { + "id": "2606.24515", + "title": "Reinforcement Learning for Computer-Use Agents with Autonomous Evaluation", + "url": "https://arxiv.org/abs/2606.24515", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Marta Sumyk", + "Oleksandr Kosovan" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.24515", + "source": "arxiv", + "source_id": "arxiv:2606.24515", + "pdf_url": "https://arxiv.org/pdf/2606.24515", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.24694", + "title": "SupplyNet: Supporting Visual Exploratory Learning in Supply Chain via Contextual Multi-Agent Simulation", + "url": "https://arxiv.org/abs/2606.24694", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yanjia Li", + "Kelcy Kexin Han", + "Tianrui Hu", + "Yi-Fan Cao", + "Huamin Qu", + "Sicheng Song" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "multi-agent", + "rag", + "reasoning", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24694", + "source": "arxiv", + "source_id": "arxiv:2606.24694", + "pdf_url": "https://arxiv.org/pdf/2606.24694", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24976", + "title": "Diagnosing and Mitigating Compounding Failures in Agentic Persuasion via Taxonomic Strategy Retrieval", + "url": "https://arxiv.org/abs/2606.24976", + "published": "2026-06-23", + "updated": "2026-07-04", + "authors": [ + "Sana Ayromlou", + "Purvi Sehgal", + "Pradyumna Narayana" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.24976", + "source": "arxiv", + "source_id": "arxiv:2606.24976", + "pdf_url": "https://arxiv.org/pdf/2606.24976", + "primary_query": "rag-agent" + }, + { + "id": "2606.22737", + "title": "GroundEval: A Deterministic Replacement for LLM-as-Judge in Stateful Agent Evaluation", + "url": "https://arxiv.org/abs/2606.22737", + "published": "2026-06-22", + "updated": "2026-07-02", + "authors": [ + "Jeffrey Flynt" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.22737", + "source": "arxiv", + "source_id": "arxiv:2606.22737", + "pdf_url": "https://arxiv.org/pdf/2606.22737", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.23283", + "title": "Towards Root Memories: Benchmarking and Enhancing Implicit Logical Memory Retrieval for Personalized LLMs", + "url": "https://arxiv.org/abs/2606.23283", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Hongxun Ding", + "Xiang Yu", + "Chengbing Wang", + "Jianfei Xiao", + "Keqin Bao", + "Wenjie Wang", + "Xiangnan He" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.23283", + "source": "arxiv", + "source_id": "arxiv:2606.23283", + "pdf_url": "https://arxiv.org/pdf/2606.23283", + "primary_query": "agent-memory" + }, + { + "id": "2606.23195", + "title": "Memory Contagion: Cross-Temporal Propagation of Evaluator Bias via Agent Memory", + "url": "https://arxiv.org/abs/2606.23195", + "published": "2026-06-22", + "updated": "2026-06-24", + "authors": [ + "Zewen Liu" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.23195", + "source": "arxiv", + "source_id": "arxiv:2606.23195", + "pdf_url": "https://arxiv.org/pdf/2606.23195", + "primary_query": "agent-memory" + }, + { + "id": "2606.22864", + "title": "When AUC 0.998 Is Not Enough: A Candidate Evaluation Protocol for Hidden-State Probes of Indirect Prompt Injection in Multimodal Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.22864", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Yanhang Li", + "Zhichao Fan", + "Zexin Zhuang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.22864", + "source": "arxiv", + "source_id": "arxiv:2606.22864", + "pdf_url": "https://arxiv.org/pdf/2606.22864", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.22495", + "title": "Grounded Scaling: Why Agentic AI Needs Deterministic Environments", + "url": "https://arxiv.org/abs/2606.22495", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Liang Ding", + "Xintong Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "embodied-agent", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.22495", + "source": "arxiv", + "source_id": "arxiv:2606.22495", + "pdf_url": "https://arxiv.org/pdf/2606.22495", + "primary_query": "agentic-ai" + }, + { + "id": "2606.22484", + "title": "Governed AI-Assisted Engineering: Graduated Human Oversight for Agentic Code Generation in Regulated Domains", + "url": "https://arxiv.org/abs/2606.22484", + "published": "2026-06-21", + "updated": "2026-07-04", + "authors": [ + "Richard Kang" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.22484", + "source": "arxiv", + "source_id": "arxiv:2606.22484", + "pdf_url": "https://arxiv.org/pdf/2606.22484", + "primary_query": "agentic-ai" + }, + { + "id": "2606.22030", + "title": "Nous: A Predictive World Model for Long-Term Agent Memory", + "url": "https://arxiv.org/abs/2606.22030", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Pranav Singh" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.IR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.22030", + "source": "arxiv", + "source_id": "arxiv:2606.22030", + "pdf_url": "https://arxiv.org/pdf/2606.22030", + "primary_query": "agent-memory" + }, + { + "id": "2606.22151", + "title": "Novelty-Aware Agentic Retrieval: Comparing Research Contributions Through Structured Multi-Step Reasoning", + "url": "https://arxiv.org/abs/2606.22151", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Shou-Tzu Han" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.22151", + "source": "arxiv", + "source_id": "arxiv:2606.22151", + "pdf_url": "https://arxiv.org/pdf/2606.22151", + "primary_query": "rag-agent" + }, + { + "id": "2606.21842", + "title": "Agent-Assisted Side-Channel Attacks on Non-Prefix KV Cache in RAG", + "url": "https://arxiv.org/abs/2606.21842", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "He Sun", + "Shinan Liu", + "Siyuan Ma", + "Junhao Li", + "Mingjun Xiao", + "Wenhao Jiang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.21842", + "source": "arxiv", + "source_id": "arxiv:2606.21842", + "pdf_url": "https://arxiv.org/pdf/2606.21842", + "primary_query": "rag-agent" + }, + { + "id": "2606.21409", + "title": "Don't Blindly Trust It: How Unreliable Feedback Breaks Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.21409", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Chubin Zhang", + "Zhenglin Wan", + "Xingrui Yu", + "Pengfei Zhou", + "Wangbo Zhao", + "Jingxuan Wu", + "Yaxin Zhou", + "Ivor Tsang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.21409", + "source": "arxiv", + "source_id": "arxiv:2606.21409", + "pdf_url": "https://arxiv.org/pdf/2606.21409", + "primary_query": "tool-use" + }, + { + "id": "2606.21553", + "title": "Dissecting Agentic RAG: A Component Ablation for Multi-Hop QA with a Local 7B Model", + "url": "https://arxiv.org/abs/2606.21553", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Sheroz Shaikh" + ], + "categories": [ + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.21553", + "source": "arxiv", + "source_id": "arxiv:2606.21553", + "pdf_url": "https://arxiv.org/pdf/2606.21553", + "primary_query": "rag-agent" + }, + { + "id": "2606.20470", + "title": "Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems", + "url": "https://arxiv.org/abs/2606.20470", + "published": "2026-06-18", + "updated": "2026-06-26", + "authors": [ + "Reza Soosahabi", + "Vivek Namsani" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.20470", + "source": "arxiv", + "source_id": "arxiv:2606.20470", + "pdf_url": "https://arxiv.org/pdf/2606.20470", + "primary_query": "agentic-ai" + }, + { + "id": "2606.19812", + "title": "Human-on-the-Loop Orchestration for AI-Assisted Legal Discovery", + "url": "https://arxiv.org/abs/2606.19812", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Anushree Sinha", + "Srivaths Ranganathan", + "Abhishek Dharmaratnakar", + "Debanshu Das" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "workflow-agent", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "planning-agent" + ], + "arxiv_id": "2606.19812", + "source": "arxiv", + "source_id": "arxiv:2606.19812", + "pdf_url": "https://arxiv.org/pdf/2606.19812", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20515", + "title": "S-Agent: Spatial Tool-Use Elicits Reasoning for Spatial Intelligence", + "url": "https://arxiv.org/abs/2606.20515", + "published": "2026-06-18", + "updated": "2026-06-28", + "authors": [ + "Yalun Dai", + "Hao Li", + "Shulin Tian", + "Runmao Yao", + "Yuhao Dong", + "Fangzhou Hong", + "Zhaoxi Chen", + "Fangfu Liu", + "Baoliang Tian", + "Dingwen Zhang", + "Tao Wang", + "Kim-Hui Yap", + "Ziwei Liu" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "tool-use" + ], + "arxiv_id": "2606.20515", + "source": "arxiv", + "source_id": "arxiv:2606.20515", + "pdf_url": "https://arxiv.org/pdf/2606.20515", + "primary_query": "agent-memory" + }, + { + "id": "2606.20023", + "title": "When Lower Privileges Suffice: Investigating Over-Privileged Tool Selection in LLM Agents", + "url": "https://arxiv.org/abs/2606.20023", + "published": "2026-06-18", + "updated": "2026-07-07", + "authors": [ + "Kaiyue Yang", + "Yuyan Bu", + "Jingwei Yi", + "Yuchi Wang", + "Biyu Zhou", + "Juntao Dai", + "Songlin Hu", + "Yaodong Yang" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.20023", + "source": "arxiv", + "source_id": "arxiv:2606.20023", + "pdf_url": "https://arxiv.org/pdf/2606.20023", + "primary_query": "tool-use" + }, + { + "id": "2606.20785", + "title": "Fara-1.5: Scalable Learning Environments for Computer Use Agents", + "url": "https://arxiv.org/abs/2606.20785", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Ahmed Awadallah", + "Sahil Gupta", + "Yash Lara", + "Yadong Lu", + "Hussein Mozannar", + "Akshay Nambi", + "Zach Nussbaum", + "Yash Pandya", + "Aravind Rajeswaran", + "Corby Rosset", + "Alexey Taymanov", + "Luiz do Valle", + "Vibhav Vineet", + "Spencer Whitehead", + "Andrew Zhao" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20785", + "source": "arxiv", + "source_id": "arxiv:2606.20785", + "pdf_url": "https://arxiv.org/pdf/2606.20785", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.19930", + "title": "MobileForge: Annotation-Free Adaptation for Mobile GUI Agents with Hierarchical Feedback-Guided Policy Optimization", + "url": "https://arxiv.org/abs/2606.19930", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Guangyi Liu", + "Pengxiang Zhao", + "Gao Wu", + "Yiwen Yin", + "Mading Li", + "Liang Liu", + "Congxiao Liu", + "Zhang Qi", + "Mengyan Wang", + "Liang Guo", + "Yong Liu" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.19930", + "source": "arxiv", + "source_id": "arxiv:2606.19930", + "pdf_url": "https://arxiv.org/pdf/2606.19930", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.18671", + "title": "HANSEL: Extracting Breadcrumbs from Web Agent Trajectories for Interactive Verification", + "url": "https://arxiv.org/abs/2606.18671", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Yujin Zhang", + "Daye Nam" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "planning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "web-gui-agent" + ], + "arxiv_id": "2606.18671", + "source": "arxiv", + "source_id": "arxiv:2606.18671", + "pdf_url": "https://arxiv.org/pdf/2606.18671", + "primary_query": "ai-agent" + }, + { + "id": "2606.19063", + "title": "PYPILINE: Malicious PyPI Package Detection via Suspicious API Knowledge and Agent Workflow", + "url": "https://arxiv.org/abs/2606.19063", + "published": "2026-06-17", + "updated": "2026-06-26", + "authors": [ + "Siyuan Pang", + "Yepeng Yao", + "Zhengwei Jiang", + "Zijing Fan", + "Haozhe Li", + "Baoxu Liu" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "rag-agent" + ], + "arxiv_id": "2606.19063", + "source": "arxiv", + "source_id": "arxiv:2606.19063", + "pdf_url": "https://arxiv.org/pdf/2606.19063", + "primary_query": "agentic-ai" + }, + { + "id": "2606.19409", + "title": "OpenRath: Session-Centered Runtime State for Agent Systems", + "url": "https://arxiv.org/abs/2606.19409", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Fukang Wen", + "Zhijie Wang", + "Ruilin Xu" + ], + "categories": [ + "cs.SE", + "cs.PL" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.19409", + "source": "arxiv", + "source_id": "arxiv:2606.19409", + "pdf_url": "https://arxiv.org/pdf/2606.19409", + "primary_query": "agent-memory" + }, + { + "id": "2606.19613", + "title": "StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns", + "url": "https://arxiv.org/abs/2606.19613", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Vlad Sobal", + "Shuo Yang", + "Yuting Zhang", + "Wei Xia", + "Stefano Soatto" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.19613", + "source": "arxiv", + "source_id": "arxiv:2606.19613", + "pdf_url": "https://arxiv.org/pdf/2606.19613", + "primary_query": "coding-agent" + }, + { + "id": "2606.20717", + "title": "MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents", + "url": "https://arxiv.org/abs/2606.20717", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Xuelong Dai", + "Jianyu Ma", + "Boyang Ma", + "Biwei Yan", + "Yijun Yang", + "Yue Zhang" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20717", + "source": "arxiv", + "source_id": "arxiv:2606.20717", + "pdf_url": "https://arxiv.org/pdf/2606.20717", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.16111", + "title": "Towards Pareto-Optimal Tool-Integrated Agents with Pareto Ranking Policy Optimization", + "url": "https://arxiv.org/abs/2606.16111", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Junyi Li", + "Xiaowei Qian", + "Yingyi Zhang", + "Wenlin Zhang", + "Guojing Li", + "Sheng Zhang", + "Xiao Han", + "Yichao Wang", + "Xiangyu Zhao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-safety", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent", + "tool-use" + ], + "arxiv_id": "2606.16111", + "source": "arxiv", + "source_id": "arxiv:2606.16111", + "pdf_url": "https://arxiv.org/pdf/2606.16111", + "primary_query": "language-agent" + }, + { + "id": "2606.16748", + "title": "MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.16748", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Lawrence Keunho Jang", + "Andrew Keunwoo Jang", + "Jing Yu Koh", + "Ruslan Salakhutdinov" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "web-gui-agent" + ], + "arxiv_id": "2606.16748", + "source": "arxiv", + "source_id": "arxiv:2606.16748", + "pdf_url": "https://arxiv.org/pdf/2606.16748", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.15591", + "title": "Agentic Retrieval and Reinforcement Learned Equation Chains: A Controlled Generation Framework for Complex and Novel Physics Word Problems", + "url": "https://arxiv.org/abs/2606.15591", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Tirthankar Mittra" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.15591", + "source": "arxiv", + "source_id": "arxiv:2606.15591", + "pdf_url": "https://arxiv.org/pdf/2606.15591", + "primary_query": "rag-agent" + }, + { + "id": "2606.15152", + "title": "Can Agents Read the Room? Benchmarking Visual Social Intelligence in Multimodal Simulation", + "url": "https://arxiv.org/abs/2606.15152", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Shijun Wan", + "Xuehai Wu", + "Jiwen Zhang", + "Siyuan Wang", + "Zhongyu Wei" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.15152", + "source": "arxiv", + "source_id": "arxiv:2606.15152", + "pdf_url": "https://arxiv.org/pdf/2606.15152", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.15242", + "title": "Benign in Isolation, Harmful in Composition: Security Risks in Agent Skill Ecosystems", + "url": "https://arxiv.org/abs/2606.15242", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Yi Xie", + "Jiawei Du", + "Yu Cheng", + "Jiuan Zhou", + "Zhaoxia Yin" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.15242", + "source": "arxiv", + "source_id": "arxiv:2606.15242", + "pdf_url": "https://arxiv.org/pdf/2606.15242", + "primary_query": "planning-agent" + }, + { + "id": "2606.14502", + "title": "From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI", + "url": "https://arxiv.org/abs/2606.14502", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Yongheng Zhang", + "Ziang Liu", + "Jiaxuan Zhu", + "Shuai Wang", + "Xiangqi Chen", + "Haojing Huang", + "Jiayi Kuang", + "Siyu Chen", + "Ao Shen", + "Hao Wu", + "Qiufeng Wang", + "Qian-Wen Zhang", + "Junnan Dong", + "Wenhao Jiang", + "Ying Shen", + "Hai-Tao Zheng", + "Yinghui Li", + "Di Yin", + "Xing Sun", + "Philip S. Yu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.14502", + "source": "arxiv", + "source_id": "arxiv:2606.14502", + "pdf_url": "https://arxiv.org/pdf/2606.14502", + "primary_query": "tool-use" + }, + { + "id": "2606.15017", + "title": "Are Online Skill and Memory Modules Always Worth Their Tokens? A Budget-Constrained Study of Web Agents", + "url": "https://arxiv.org/abs/2606.15017", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Sina Hajimiri", + "Masih Aminbeidokhti", + "Jose Dolz", + "Ismail Ben Ayed", + "Issam H. Laradji", + "Spandana Gella", + "Nicolas Gontier" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.15017", + "source": "arxiv", + "source_id": "arxiv:2606.15017", + "pdf_url": "https://arxiv.org/pdf/2606.15017", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.14574", + "title": "SIMMER: Benchmarking Latent Failures in LLM Executable Planning with a World Model", + "url": "https://arxiv.org/abs/2606.14574", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Xiaoxin Lu", + "Ranran Haoran Zhang", + "Rui Zhang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.14574", + "source": "arxiv", + "source_id": "arxiv:2606.14574", + "pdf_url": "https://arxiv.org/pdf/2606.14574", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.13317", + "title": "SkillCAT: Contrastive Assessment and Topology-Aware Skill Self-Evolution for LLM Agents", + "url": "https://arxiv.org/abs/2606.13317", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Kunfeng Chen", + "Qihuang Zhong", + "Juhua Liu", + "Bo Du" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.13317", + "source": "arxiv", + "source_id": "arxiv:2606.13317", + "pdf_url": "https://arxiv.org/pdf/2606.13317", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.14805", + "title": "Knowledge-Based Zero-Replay Debugging of Multi-Agent LLM Traces", + "url": "https://arxiv.org/abs/2606.14805", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Dong Ho Kang", + "Hyeonjeong Cha", + "Daein Weon" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "memory", + "multi-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.14805", + "source": "arxiv", + "source_id": "arxiv:2606.14805", + "pdf_url": "https://arxiv.org/pdf/2606.14805", + "primary_query": "tool-use" + }, + { + "id": "2606.13602", + "title": "EpiBench: Verifiable Evaluation of AI Agents on Epigenomics Analysis", + "url": "https://arxiv.org/abs/2606.13602", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Harihara Muralidharan", + "Reema Baskar", + "Soo Hee Lee", + "Tim Proctor", + "Kenny Workman" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.13602", + "source": "arxiv", + "source_id": "arxiv:2606.13602", + "pdf_url": "https://arxiv.org/pdf/2606.13602", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13192", + "title": "Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach", + "url": "https://arxiv.org/abs/2606.13192", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Ruichao Mao", + "Zhou Fang", + "Teng Guo", + "Hao Yang", + "Yaping Li", + "Shaohua Peng", + "Maji Huang", + "Xiaoyu Lin", + "Shuoyang Liu", + "Xuepeng Li", + "Yuyu Zhang", + "Hai Rao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.13192", + "source": "arxiv", + "source_id": "arxiv:2606.13192", + "pdf_url": "https://arxiv.org/pdf/2606.13192", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.12674", + "title": "Evoflux: Inference-Time Evolution of Executable Tool Workflows for Compact Agents", + "url": "https://arxiv.org/abs/2606.12674", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Kushal Raj Bhandari", + "Ling Yue", + "Ching-Yun Ko", + "Dhaval Patel", + "Shaowu Pan", + "Pin-Yu Chen", + "Jianxi Gao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling", + "tool-use" + ], + "arxiv_id": "2606.12674", + "source": "arxiv", + "source_id": "arxiv:2606.12674", + "pdf_url": "https://arxiv.org/pdf/2606.12674", + "primary_query": "function-calling" + }, + { + "id": "2606.12384", + "title": "APPO: Agentic Procedural Policy Optimization", + "url": "https://arxiv.org/abs/2606.12384", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Xucong Wang", + "Ziyu Ma", + "Yong Wang", + "Yuxiang Ji", + "Shidong Yang", + "Guanhua Chen", + "Pengkun Wang", + "Xiangxiang Chu" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.12384", + "source": "arxiv", + "source_id": "arxiv:2606.12384", + "pdf_url": "https://arxiv.org/pdf/2606.12384", + "primary_query": "tool-use" + }, + { + "id": "2606.12563", + "title": "Arbor: Tree Search as a Cognition Layer for Autonomous Agents", + "url": "https://arxiv.org/abs/2606.12563", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Neha Prakriya", + "Chaojun Hou", + "Zheng Gong", + "Huasha Zhao", + "Xi Zhao", + "Mou Li", + "Zhenyu Gu", + "Emad Barsoum" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.12563", + "source": "arxiv", + "source_id": "arxiv:2606.12563", + "pdf_url": "https://arxiv.org/pdf/2606.12563", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.11079", + "title": "VISTA: A Versatile Interactive User Simulation Toolkit for Agent Evaluation", + "url": "https://arxiv.org/abs/2606.11079", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yunan Lu", + "Ryan Shea", + "Yusen Zhang", + "Zhou Yu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.11079", + "source": "arxiv", + "source_id": "arxiv:2606.11079", + "pdf_url": "https://arxiv.org/pdf/2606.11079", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.10742", + "title": "MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents", + "url": "https://arxiv.org/abs/2606.10742", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yv Zhang", + "Hao Sun", + "Hao Fang", + "Kuofeng Gao", + "Fan Mo", + "Bin Chen", + "Shu-Tao Xia", + "Yaowei Wang" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.10742", + "source": "arxiv", + "source_id": "arxiv:2606.10742", + "pdf_url": "https://arxiv.org/pdf/2606.10742", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09774", + "title": "Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters", + "url": "https://arxiv.org/abs/2606.09774", + "published": "2026-06-08", + "updated": "2026-06-25", + "authors": [ + "Matthew Ho", + "Brian Liu", + "Jixuan Chen", + "Audrey Wang", + "Lianhui Qin" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "rag", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.09774", + "source": "arxiv", + "source_id": "arxiv:2606.09774", + "pdf_url": "https://arxiv.org/pdf/2606.09774", + "primary_query": "agent-memory" + }, + { + "id": "2606.09198", + "title": "MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation", + "url": "https://arxiv.org/abs/2606.09198", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Yongrui Liu", + "Deyi Xiong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.09198", + "source": "arxiv", + "source_id": "arxiv:2606.09198", + "pdf_url": "https://arxiv.org/pdf/2606.09198", + "primary_query": "agent-memory" + }, + { + "id": "2606.08790", + "title": "RAILS: Verification-Native Clearing For Agentic Commerce", + "url": "https://arxiv.org/abs/2606.08790", + "published": "2026-06-07", + "updated": "2026-06-07", + "authors": [ + "Adrian de Valois-Franklin", + "Alex Bogdan" + ], + "categories": [ + "cs.AI", + "cs.CR", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.08790", + "source": "arxiv", + "source_id": "arxiv:2606.08790", + "pdf_url": "https://arxiv.org/pdf/2606.08790", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.08625", + "title": "From Holistic Evaluation to Structured Criteria: Rubrics Across the Evolving LLM Landscape", + "url": "https://arxiv.org/abs/2606.08625", + "published": "2026-06-07", + "updated": "2026-07-01", + "authors": [ + "Hao Chen", + "Ziyu Han", + "Yukun Yan", + "Qingfu Zhu", + "Maosong Sun", + "Wanxiang Che" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.08625", + "source": "arxiv", + "source_id": "arxiv:2606.08625", + "pdf_url": "https://arxiv.org/pdf/2606.08625", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.07379", + "title": "Do Coding Agents Deceive Us? Detecting and Preventing Cheating via Capped Evaluation with Randomized Tests", + "url": "https://arxiv.org/abs/2606.07379", + "published": "2026-06-05", + "updated": "2026-06-08", + "authors": [ + "Thanawat Lodkaew", + "Johannes Ackermann", + "Soichiro Nishimori", + "Nontawat Charoenphakdee", + "Masashi Sugiyama", + "Takashi Ishida" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL", + "stat.ME" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.07379", + "source": "arxiv", + "source_id": "arxiv:2606.07379", + "pdf_url": "https://arxiv.org/pdf/2606.07379", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.05548", + "title": "ADK Arena: Evaluating Agent Development Kits via LLM-as-a-Developer", + "url": "https://arxiv.org/abs/2606.05548", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Jintao Huang", + "Xiaomin Li", + "Gaurav Mittal", + "Yu Hu" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.05548", + "source": "arxiv", + "source_id": "arxiv:2606.05548", + "pdf_url": "https://arxiv.org/pdf/2606.05548", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.06473", + "title": "MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery", + "url": "https://arxiv.org/abs/2606.06473", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Shangheng Du", + "Xiangchao Yan", + "Jinxin Shi", + "Zongsheng Cao", + "Shiyang Feng", + "Zichen Liang", + "Boyuan Sun", + "Tianshuo Peng", + "Yifan Zhou", + "Xin Li", + "Jie Zhou", + "Liang He", + "Bo Zhang", + "Lei Bai" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "multi-agent", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.06473", + "source": "arxiv", + "source_id": "arxiv:2606.06473", + "pdf_url": "https://arxiv.org/pdf/2606.06473", + "primary_query": "planning-agent" + }, + { + "id": "2606.06462", + "title": "Benchmark Everything Everywhere All at Once", + "url": "https://arxiv.org/abs/2606.06462", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Shiyun Xiong", + "Dongming Wu", + "Peiwen Sun", + "Yuang Ai", + "Bokang Yang", + "Wencheng Han", + "Xiao-Hui Li", + "Xiangyu Yue" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.06462", + "source": "arxiv", + "source_id": "arxiv:2606.06462", + "pdf_url": "https://arxiv.org/pdf/2606.06462", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.05263", + "title": "Policy-Conditioned Counterfactual Credit for Verifiable Reinforcement Learning of Long-Horizon Language Agents", + "url": "https://arxiv.org/abs/2606.05263", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Renwei Meng" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.05263", + "source": "arxiv", + "source_id": "arxiv:2606.05263", + "pdf_url": "https://arxiv.org/pdf/2606.05263", + "primary_query": "language-agent" + }, + { + "id": "2606.04628", + "title": "RAMPART: Registry-based Agentic Memory with Priority-Aware Runtime Transformation", + "url": "https://arxiv.org/abs/2606.04628", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Nikodem Tomczak" + ], + "categories": [ + "cs.CL", + "cs.MA" + ], + "topics": [ + "memory" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04628", + "source": "arxiv", + "source_id": "arxiv:2606.04628", + "pdf_url": "https://arxiv.org/pdf/2606.04628", + "primary_query": "agent-memory" + }, + { + "id": "2606.05414", + "title": "When Evidence is Sparse: Weakly Supervised Early Failure Alerting in Dialogs and LLM-Agent Trajectories", + "url": "https://arxiv.org/abs/2606.05414", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Avinash Baidya", + "Xinran Liang", + "Ruocheng Guo", + "Xiang Gao", + "Kamalika Das" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.HC", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.05414", + "source": "arxiv", + "source_id": "arxiv:2606.05414", + "pdf_url": "https://arxiv.org/pdf/2606.05414", + "primary_query": "planning-agent" + }, + { + "id": "2606.03329", + "title": "InfoMem: Training Long-Context Memory Agents with Answer-Conditioned Information Gain", + "url": "https://arxiv.org/abs/2606.03329", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Tiancheng Han", + "Yong Li", + "Wuzhou Yu", + "Qiaosheng Zhang", + "Wenqi Shao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.03329", + "source": "arxiv", + "source_id": "arxiv:2606.03329", + "pdf_url": "https://arxiv.org/pdf/2606.03329", + "primary_query": "agent-memory" + }, + { + "id": "2606.02965", + "title": "What Benchmarks Don't Measure: The Case for Evaluating Abstention Competence in Autonomous Agents", + "url": "https://arxiv.org/abs/2606.02965", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Victor Ojewale", + "Suresh Venkatasubramanian" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.02965", + "source": "arxiv", + "source_id": "arxiv:2606.02965", + "pdf_url": "https://arxiv.org/pdf/2606.02965", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.02404", + "title": "K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts", + "url": "https://arxiv.org/abs/2606.02404", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Nahyun Lee", + "Dongkeun Yoon", + "Guijin Son", + "Geewook Kim", + "Dayoon Ko", + "Jeonghun Park", + "Haneul Yoo", + "Jaewon Cho", + "Junghun Park", + "Changyoon Lee", + "Kyochul Jang", + "Jaeyeon Kim", + "Eunsu Kim", + "Woojin Cho", + "Seungone Kim" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.02404", + "source": "arxiv", + "source_id": "arxiv:2606.02404", + "pdf_url": "https://arxiv.org/pdf/2606.02404", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.09863", + "title": "From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents", + "url": "https://arxiv.org/abs/2606.09863", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Laksh Advani" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.09863", + "source": "arxiv", + "source_id": "arxiv:2606.09863", + "pdf_url": "https://arxiv.org/pdf/2606.09863", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.00611", + "title": "TRACE: Trajectory Risk-Aware Compression for Long-Horizon Agent Safety", + "url": "https://arxiv.org/abs/2606.00611", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Zhepei Hong", + "Lin Wang", + "Liting Li", + "Haokai Ma", + "Junfeng Fang", + "Fei Shen", + "Dan Zhang", + "Xiang Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.00611", + "source": "arxiv", + "source_id": "arxiv:2606.00611", + "pdf_url": "https://arxiv.org/pdf/2606.00611", + "primary_query": "agent-safety" + }, + { + "id": "2606.00198", + "title": "BAGEN: Are LLM Agents Budget-Aware?", + "url": "https://arxiv.org/abs/2606.00198", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Yuxiang Lin", + "Zihan Wang", + "Mengyang Liu", + "Yuxuan Shan", + "Longju Bai", + "Junyao Zhang", + "Xing Jin", + "Boshan Chen", + "Jinyan Su", + "Xingyao Wang", + "Jiaxin Pei", + "Manling Li" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.00198", + "source": "arxiv", + "source_id": "arxiv:2606.00198", + "pdf_url": "https://arxiv.org/pdf/2606.00198", + "primary_query": "planning-agent" + }, + { + "id": "2605.29790", + "title": "Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems", + "url": "https://arxiv.org/abs/2605.29790", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Zhezheng Hao", + "Tianfu Wang", + "Huanshuo Dong", + "Ziyan Liu", + "Hong Wang", + "Xiankun Lin", + "Qiang Lin", + "Can Wang", + "Hande Dong", + "Jiawei Chen" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.29790", + "source": "arxiv", + "source_id": "arxiv:2605.29790", + "pdf_url": "https://arxiv.org/pdf/2605.29790", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.29640", + "title": "VikingMem: A Memory Base Management System for Stateful LLM-based Applications", + "url": "https://arxiv.org/abs/2605.29640", + "published": "2026-05-28", + "updated": "2026-06-12", + "authors": [ + "Jiajie Fu", + "Junwen Chen", + "Mengzhao Wang", + "Aoxiang He", + "Maojia Sheng", + "Xiangyu Ke", + "Yifan Zhu", + "Yunjun Gao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.29640", + "source": "arxiv", + "source_id": "arxiv:2605.29640", + "pdf_url": "https://arxiv.org/pdf/2605.29640", + "primary_query": "agent-memory" + }, + { + "id": "2605.29801", + "title": "AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security", + "url": "https://arxiv.org/abs/2605.29801", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Dongrui Liu", + "Yu Li", + "Zhonghao Yang", + "Peng Wang", + "Guanxu Chen", + "Yuejin Xie", + "Qinghua Mao", + "Wanying Qu", + "Yanxu Zhu", + "Tianyi Zhou", + "Leitao Yuan", + "Zhijie Zheng", + "Qihao Lin", + "Yimin Wang", + "Haoyu Luo", + "Shuai Shao", + "Chen Qian", + "Qingyu Liu", + "Ling Tang", + "Ruiyang Qin", + "Qihan Ren", + "Junxiao Yang", + "Kun Wang", + "Zhiheng Xi", + "Linfeng Zhang", + "Ranjie Duan", + "Bo Zhang", + "Wenjie Wang", + "Wen Shen", + "Qiaosheng Zhang", + "Yan Teng", + "Chaochao Lu", + "Rui Mei", + "Man Li", + "Jialing Tao", + "Xi Lin", + "Tianhang Zheng", + "Yong Liu", + "Quanshi Zhang", + "Lei Zhu", + "Xingjun Ma", + "Junhua Liu", + "Hui Xue", + "Xiaoxiang Zuo", + "Xiangnan He", + "Chao Shen", + "Xianglong Liu", + "Minlie Huang", + "Jing Shao", + "Xia Hu" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CR", + "cs.CV", + "cs.LG" + ], + "topics": [ + "agent-safety", + "computer-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.29801", + "source": "arxiv", + "source_id": "arxiv:2605.29801", + "pdf_url": "https://arxiv.org/pdf/2605.29801", + "primary_query": "agent-safety" + }, + { + "id": "2605.30407", + "title": "Exploring Autonomous Agentic Data Engineering for Model Specialization", + "url": "https://arxiv.org/abs/2605.30407", + "published": "2026-05-28", + "updated": "2026-06-08", + "authors": [ + "Yujie Luo", + "Xiangyuan Ru", + "Jingsheng Zheng", + "Jingjing Wang", + "Yuqi Zhu", + "Jintian Zhang", + "Runnan Fang", + "Kewei Xu", + "Ye Liu", + "Zheng Wei", + "Jiang Bian", + "Zang Li", + "Shumin Deng" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.30407", + "source": "arxiv", + "source_id": "arxiv:2605.30407", + "pdf_url": "https://arxiv.org/pdf/2605.30407", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.27825", + "title": "MRMMIA: Membership Inference Attacks on Memory in Chat Agents", + "url": "https://arxiv.org/abs/2605.27825", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Kai Chen", + "Yan Pang", + "Tianhao Wang" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.27825", + "source": "arxiv", + "source_id": "arxiv:2605.27825", + "pdf_url": "https://arxiv.org/pdf/2605.27825", + "primary_query": "agent-memory" + }, + { + "id": "2605.28175", + "title": "Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation", + "url": "https://arxiv.org/abs/2605.28175", + "published": "2026-05-27", + "updated": "2026-05-29", + "authors": [ + "Shijie Wang", + "Chengyi Liu", + "Yujuan Ding", + "Shanru Lin", + "See-Kiong Ng", + "Xu Xin", + "Wenqi Fan" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.28175", + "source": "arxiv", + "source_id": "arxiv:2605.28175", + "pdf_url": "https://arxiv.org/pdf/2605.28175", + "primary_query": "rag-agent" + }, + { + "id": "2605.27935", + "title": "Do Agents Think Deeper? A Mechanistic Investigation of Layer-Wise Dynamics in Sequential Planning", + "url": "https://arxiv.org/abs/2605.27935", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Zhenyu Cui", + "Xiangzhong Luo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2605.27935", + "source": "arxiv", + "source_id": "arxiv:2605.27935", + "pdf_url": "https://arxiv.org/pdf/2605.27935", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.25920", + "title": "Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning", + "url": "https://arxiv.org/abs/2605.25920", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Wei Fan", + "Yining Zhou", + "Mufan Zhang", + "Yanbing Weng", + "Yiran HU", + "Tianshi Zheng", + "Baixuan Xu", + "Chunyang Li", + "Jianhui Yang", + "Haoran Li", + "Yangqiu Song" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.25920", + "source": "arxiv", + "source_id": "arxiv:2605.25920", + "pdf_url": "https://arxiv.org/pdf/2605.25920", + "primary_query": "rag-agent" + }, + { + "id": "2605.25393", + "title": "Decision-Making with Lightweight Confidence-Aware Language Model for Autonomous Driving", + "url": "https://arxiv.org/abs/2605.25393", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Ruoyu Yao", + "Ruiguo Zhong", + "Pei Liu", + "Mingxing Peng", + "Rui Yang", + "Jun Ma" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.25393", + "source": "arxiv", + "source_id": "arxiv:2605.25393", + "pdf_url": "https://arxiv.org/pdf/2605.25393", + "primary_query": "rag-agent" + }, + { + "id": "2605.25310", + "title": "Tool-Call Dependency Structure is Linearly Decodable in LLM Agent Residual Streams", + "url": "https://arxiv.org/abs/2605.25310", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Tianda Sun", + "Dimitar Kazakov" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.25310", + "source": "arxiv", + "source_id": "arxiv:2605.25310", + "pdf_url": "https://arxiv.org/pdf/2605.25310", + "primary_query": "planning-agent" + }, + { + "id": "2605.24309", + "title": "Reframing LLM Agent Security as an Agent-Human Interaction Problem", + "url": "https://arxiv.org/abs/2605.24309", + "published": "2026-05-23", + "updated": "2026-05-23", + "authors": [ + "Peiran Wang", + "Ying Li", + "Yuan Tian" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.24309", + "source": "arxiv", + "source_id": "arxiv:2605.24309", + "pdf_url": "https://arxiv.org/pdf/2605.24309", + "primary_query": "agent-safety" + }, + { + "id": "2605.19604", + "title": "Formal Skill: Programmable Runtime Skills for Efficient and Accurate LLM Agents", + "url": "https://arxiv.org/abs/2605.19604", + "published": "2026-05-19", + "updated": "2026-05-19", + "authors": [ + "Xi Zhang", + "Meijun Gao", + "Yuntian Zhao", + "Xinyu Tan", + "Yilun Yao", + "Feiyu Wang", + "Yanshu Wang", + "Dingsiyi", + "Tong Yang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.19604", + "source": "arxiv", + "source_id": "arxiv:2605.19604", + "pdf_url": "https://arxiv.org/pdf/2605.19604", + "primary_query": "function-calling" + }, + { + "id": "2605.18502", + "title": "The distance-based formation controller design for multi-agent systems in port-Hamiltonian form", + "url": "https://arxiv.org/abs/2605.18502", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Jingyi Zhao", + "Yongxin Wu", + "Héctor García de Marina", + "Yuhu Wu", + "Yann Le Gorrec" + ], + "categories": [ + "math.OC" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.18502", + "source": "arxiv", + "source_id": "arxiv:2605.18502", + "pdf_url": "https://arxiv.org/pdf/2605.18502", + "primary_query": "agent-safety" + }, + { + "id": "2605.15701", + "title": "H-Mem: A Novel Memory Mechanism for Evolving and Retrieving Agent Memory via a Hybrid Structure", + "url": "https://arxiv.org/abs/2605.15701", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Jiawei Yu", + "Yixiang Fang", + "Xilin Liu", + "Yuchi Ma" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.15701", + "source": "arxiv", + "source_id": "arxiv:2605.15701", + "pdf_url": "https://arxiv.org/pdf/2605.15701", + "primary_query": "agent-memory" + }, + { + "id": "2605.15625", + "title": "ColPackAgent: Agent-Skill-Guided Hard-Particle Monte Carlo Workflows for Colloidal Packing", + "url": "https://arxiv.org/abs/2605.15625", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Lijie Ding", + "Changwoo Do" + ], + "categories": [ + "cs.AI", + "cond-mat.soft" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.15625", + "source": "arxiv", + "source_id": "arxiv:2605.15625", + "pdf_url": "https://arxiv.org/pdf/2605.15625", + "primary_query": "planning-agent" + }, + { + "id": "2605.14290", + "title": "Web Agents Should Adopt the Plan-Then-Execute Paradigm", + "url": "https://arxiv.org/abs/2605.14290", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Julien Piet", + "Annabella Chow", + "Yiwei Hou", + "Muxi Lyu", + "Sylvie Venuto", + "Jinhao Zhu", + "Raluca Ada Popa", + "David Wagner" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.14290", + "source": "arxiv", + "source_id": "arxiv:2605.14290", + "pdf_url": "https://arxiv.org/pdf/2605.14290", + "primary_query": "planning-agent" + }, + { + "id": "2605.13716", + "title": "SkillOps: Managing LLM Agent Skill Libraries as Self-Maintaining Software Ecosystems", + "url": "https://arxiv.org/abs/2605.13716", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Hongji Pu", + "Xinyuan Song", + "Liang Zhao" + ], + "categories": [ + "cs.SE", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.13716", + "source": "arxiv", + "source_id": "arxiv:2605.13716", + "pdf_url": "https://arxiv.org/pdf/2605.13716", + "primary_query": "planning-agent" + }, + { + "id": "2605.13618", + "title": "OpenAaaS: An Open Agent-as-a-Service Framework for Distributed Materials-Informatics Research", + "url": "https://arxiv.org/abs/2605.13618", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Peng Kang", + "Bixuan Li", + "Xiaoya Huang", + "Shuo Shi", + "Weiqiao Zhou", + "Zhen Li", + "Yu Liu", + "Lei Zheng" + ], + "categories": [ + "cond-mat.mtrl-sci", + "cs.AI" + ], + "topics": [ + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.13618", + "source": "arxiv", + "source_id": "arxiv:2605.13618", + "pdf_url": "https://arxiv.org/pdf/2605.13618", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.07112", + "title": "Switchcraft: AI Model Router for Agentic Tool Calling", + "url": "https://arxiv.org/abs/2605.07112", + "published": "2026-05-08", + "updated": "2026-05-08", + "authors": [ + "Sharad Agarwal", + "Pooria Namyar", + "Alec Wolman", + "Rahul Ambavat", + "Ankur Gupta", + "Qizheng Zhang" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.07112", + "source": "arxiv", + "source_id": "arxiv:2605.07112", + "pdf_url": "https://arxiv.org/pdf/2605.07112", + "primary_query": "function-calling" + }, + { + "id": "2605.06992", + "title": "Why Does Agentic Safety Fail to Generalize Across Tasks?", + "url": "https://arxiv.org/abs/2605.06992", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Yonatan Slutzky", + "Yotam Alexander", + "Tomer Slor", + "Yoav Nagel", + "Nadav Cohen" + ], + "categories": [ + "cs.LG", + "stat.ML" + ], + "topics": [ + "agent-safety", + "embodied-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.06992", + "source": "arxiv", + "source_id": "arxiv:2605.06992", + "pdf_url": "https://arxiv.org/pdf/2605.06992", + "primary_query": "agent-safety" + }, + { + "id": "2605.06957", + "title": "Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents", + "url": "https://arxiv.org/abs/2605.06957", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Shirin Sohrabi", + "Haritha Ananthakrishnan", + "Harsha Kokel", + "Kavitha Srinivas", + "Michael Katz" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.06957", + "source": "arxiv", + "source_id": "arxiv:2605.06957", + "pdf_url": "https://arxiv.org/pdf/2605.06957", + "primary_query": "planning-agent" + }, + { + "id": "2605.06737", + "title": "A Self-Healing Framework for Reliable LLM-Based Autonomous Agents", + "url": "https://arxiv.org/abs/2605.06737", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Cheonsu Jeong", + "Younggun Shin" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.06737", + "source": "arxiv", + "source_id": "arxiv:2605.06737", + "pdf_url": "https://arxiv.org/pdf/2605.06737", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.27464", + "title": "Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study", + "url": "https://arxiv.org/abs/2604.27464", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Luyao Xu", + "Xiang Chen" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27464", + "source": "arxiv", + "source_id": "arxiv:2604.27464", + "pdf_url": "https://arxiv.org/pdf/2604.27464", + "primary_query": "agent-safety" + }, + { + "id": "2604.28157", + "title": "FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption", + "url": "https://arxiv.org/abs/2604.28157", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Yanting Wang", + "Chenlong Yin", + "Ying Chen", + "Jinyuan Jia" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.28157", + "source": "arxiv", + "source_id": "arxiv:2604.28157", + "pdf_url": "https://arxiv.org/pdf/2604.28157", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.27859", + "title": "Rethinking Agentic Reinforcement Learning In Large Language Models", + "url": "https://arxiv.org/abs/2604.27859", + "published": "2026-04-30", + "updated": "2026-05-15", + "authors": [ + "Fangming Cui", + "Ruixiao Zhu", + "Cheng Fang", + "Sunan Li", + "Jiahong Li" + ], + "categories": [ + "cs.AI", + "cs.ET" + ], + "topics": [ + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27859", + "source": "arxiv", + "source_id": "arxiv:2604.27859", + "pdf_url": "https://arxiv.org/pdf/2604.27859", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.20573", + "title": "AONA: A Comprehensive Architecture and Workflow Design for Global Agentic Collaboration", + "url": "https://arxiv.org/abs/2606.20573", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Jinliang Xu", + "Runkai Zhu", + "Bingqi Li", + "Fanjie Nie", + "Jin Li", + "Jiagui Xie" + ], + "categories": [ + "cs.NI", + "cs.MA" + ], + "topics": [ + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.20573", + "source": "arxiv", + "source_id": "arxiv:2606.20573", + "pdf_url": "https://arxiv.org/pdf/2606.20573", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.25555", + "title": "From CRUD to Autonomous Agents: Formal Validation and Zero-Trust Security for Semantic Gateways in AI-Native Enterprise Systems", + "url": "https://arxiv.org/abs/2604.25555", + "published": "2026-04-28", + "updated": "2026-04-28", + "authors": [ + "Ignacio Peyrano" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.25555", + "source": "arxiv", + "source_id": "arxiv:2604.25555", + "pdf_url": "https://arxiv.org/pdf/2604.25555", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.21190", + "title": "SpatiO: Adaptive Test-Time Orchestration of Vision-Language Agents for Spatial Reasoning", + "url": "https://arxiv.org/abs/2604.21190", + "published": "2026-04-23", + "updated": "2026-06-27", + "authors": [ + "Chan Yeong Hwang", + "Miso Choi", + "Sunghyun On", + "Jinkyu Kim", + "Jungbeom Lee" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.21190", + "source": "arxiv", + "source_id": "arxiv:2604.21190", + "pdf_url": "https://arxiv.org/pdf/2604.21190", + "primary_query": "language-agent" + }, + { + "id": "2604.16706", + "title": "Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench", + "url": "https://arxiv.org/abs/2604.16706", + "published": "2026-04-17", + "updated": "2026-04-17", + "authors": [ + "Bhaskar Gurram" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.16706", + "source": "arxiv", + "source_id": "arxiv:2604.16706", + "pdf_url": "https://arxiv.org/pdf/2604.16706", + "primary_query": "language-agent" + }, + { + "id": "2604.15579", + "title": "Don't Make Models Guess Security and Safety: Symbolic Guardrails for Domain-Specific AI Agents", + "url": "https://arxiv.org/abs/2604.15579", + "published": "2026-04-16", + "updated": "2026-07-05", + "authors": [ + "Yining Hong", + "Yining She", + "Eunsuk Kang", + "Christopher S. Timperley", + "Christian Kästner" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.15579", + "source": "arxiv", + "source_id": "arxiv:2604.15579", + "pdf_url": "https://arxiv.org/pdf/2604.15579", + "primary_query": "agent-safety" + }, + { + "id": "2604.15415", + "title": "HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?", + "url": "https://arxiv.org/abs/2604.15415", + "published": "2026-04-16", + "updated": "2026-04-16", + "authors": [ + "Yukun Jiang", + "Yage Zhang", + "Michael Backes", + "Xinyue Shen", + "Yang Zhang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.15415", + "source": "arxiv", + "source_id": "arxiv:2604.15415", + "pdf_url": "https://arxiv.org/pdf/2604.15415", + "primary_query": "agent-safety" + }, + { + "id": "2604.14399", + "title": "SpaceMind: A Modular and Self-Evolving Embodied Vision-Language Agent Framework for Autonomous On-orbit Servicing", + "url": "https://arxiv.org/abs/2604.14399", + "published": "2026-04-15", + "updated": "2026-04-15", + "authors": [ + "Aodi Wu", + "Haodong Han", + "Xubo Luo", + "Ruisuo Wang", + "Shan He", + "Xue Wan" + ], + "categories": [ + "cs.RO", + "cs.AI", + "eess.SY" + ], + "topics": [ + "embodied-agent", + "reasoning", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.14399", + "source": "arxiv", + "source_id": "arxiv:2604.14399", + "pdf_url": "https://arxiv.org/pdf/2604.14399", + "primary_query": "language-agent" + }, + { + "id": "2604.08388", + "title": "Awakening the Sleeping Agent: Lean-Specific Agentic Data Reactivates General Tool Use in Goedel Prover", + "url": "https://arxiv.org/abs/2604.08388", + "published": "2026-04-09", + "updated": "2026-04-09", + "authors": [ + "Jui-Hui Chung", + "Hongzhou Lin", + "Lai Jiang", + "Shange Tang", + "Chi Jin" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2604.08388", + "source": "arxiv", + "source_id": "arxiv:2604.08388", + "pdf_url": "https://arxiv.org/pdf/2604.08388", + "primary_query": "function-calling" + }, + { + "id": "2604.06762", + "title": "ARuleCon: Agentic Security Rule Conversion", + "url": "https://arxiv.org/abs/2604.06762", + "published": "2026-04-08", + "updated": "2026-04-08", + "authors": [ + "Ming Xu", + "Hongtai Wang", + "Yanpei Guo", + "Zhengmin Yu", + "Weili Han", + "Hoon Wei Lim", + "Jin Song Dong", + "Jiaheng Zhang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.06762", + "source": "arxiv", + "source_id": "arxiv:2604.06762", + "pdf_url": "https://arxiv.org/pdf/2604.06762", + "primary_query": "agent-safety" + }, + { + "id": "2604.02155", + "title": "Brief Is Better: Non-Monotonic Chain-of-Thought Budget Effects in Function-Calling Language Agents", + "url": "https://arxiv.org/abs/2604.02155", + "published": "2026-04-02", + "updated": "2026-04-02", + "authors": [ + "Xuan Qi" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling", + "language-agent" + ], + "arxiv_id": "2604.02155", + "source": "arxiv", + "source_id": "arxiv:2604.02155", + "pdf_url": "https://arxiv.org/pdf/2604.02155", + "primary_query": "function-calling" + }, + { + "id": "2603.27148", + "title": "SafetyDrift: Predicting When AI Agents Cross the Line Before They Actually Do", + "url": "https://arxiv.org/abs/2603.27148", + "published": "2026-03-28", + "updated": "2026-03-28", + "authors": [ + "Aditya Dhodapkar", + "Farhaan Pishori" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.27148", + "source": "arxiv", + "source_id": "arxiv:2603.27148", + "pdf_url": "https://arxiv.org/pdf/2603.27148", + "primary_query": "agent-safety" + }, + { + "id": "2603.19469", + "title": "A Framework for Formalizing LLM Agent Security", + "url": "https://arxiv.org/abs/2603.19469", + "published": "2026-03-19", + "updated": "2026-03-19", + "authors": [ + "Vincent Siu", + "Jingxuan He", + "Kyle Montgomery", + "Zhun Wang", + "Neil Gong", + "Chenguang Wang", + "Dawn Song" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "memory", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.19469", + "source": "arxiv", + "source_id": "arxiv:2603.19469", + "pdf_url": "https://arxiv.org/pdf/2603.19469", + "primary_query": "agent-safety" + }, + { + "id": "2603.11088", + "title": "The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey", + "url": "https://arxiv.org/abs/2603.11088", + "published": "2026-03-11", + "updated": "2026-03-11", + "authors": [ + "Juhee Kim", + "Xiaoyuan Liu", + "Zhun Wang", + "Shi Qiu", + "Bo Li", + "Wenbo Guo", + "Dawn Song" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.11088", + "source": "arxiv", + "source_id": "arxiv:2603.11088", + "pdf_url": "https://arxiv.org/pdf/2603.11088", + "primary_query": "agent-safety" + }, + { + "id": "2603.01438", + "title": "Enhancing Persona Following at Decoding Time via Dynamic Importance Estimation for Role-Playing Agents", + "url": "https://arxiv.org/abs/2603.01438", + "published": "2026-03-02", + "updated": "2026-03-02", + "authors": [ + "Yuxin Liu", + "Mingye Zhu", + "Siyuan Liu", + "Bo Hu", + "Lei Zhang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "planning", + "rag", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.01438", + "source": "arxiv", + "source_id": "arxiv:2603.01438", + "pdf_url": "https://arxiv.org/pdf/2603.01438", + "primary_query": "language-agent" + }, + { + "id": "2602.23320", + "title": "ParamMem: Augmenting Language Agents with Parametric Reflective Memory", + "url": "https://arxiv.org/abs/2602.23320", + "published": "2026-02-26", + "updated": "2026-02-27", + "authors": [ + "Tianjun Yao", + "Yongqiang Chen", + "Yujia Zheng", + "Pan Li", + "Zhiqiang Shen", + "Kun Zhang" + ], + "categories": [ + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.23320", + "source": "arxiv", + "source_id": "arxiv:2602.23320", + "pdf_url": "https://arxiv.org/pdf/2602.23320", + "primary_query": "language-agent" + }, + { + "id": "2602.16931", + "title": "Narrow Fine-Tuning Erodes Safety Alignment in Vision-Language Agents", + "url": "https://arxiv.org/abs/2602.16931", + "published": "2026-02-18", + "updated": "2026-03-15", + "authors": [ + "Idhant Gulati", + "Shivam Raval" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.16931", + "source": "arxiv", + "source_id": "arxiv:2602.16931", + "pdf_url": "https://arxiv.org/pdf/2602.16931", + "primary_query": "language-agent" + }, + { + "id": "2602.07391", + "title": "NAAMSE: Framework for Evolutionary Security Evaluation of Agents", + "url": "https://arxiv.org/abs/2602.07391", + "published": "2026-02-07", + "updated": "2026-03-08", + "authors": [ + "Kunal Pai", + "Parth Shah", + "Harshil Patel" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.07391", + "source": "arxiv", + "source_id": "arxiv:2602.07391", + "pdf_url": "https://arxiv.org/pdf/2602.07391", + "primary_query": "agent-safety" + }, + { + "id": "2601.05467", + "title": "STELP: Secure Transpilation and Execution of LLM-Generated Programs", + "url": "https://arxiv.org/abs/2601.05467", + "published": "2026-01-09", + "updated": "2026-01-15", + "authors": [ + "Swapnil Shinde", + "Sahil Wadhwa", + "Andy Luo", + "Akshay Gupta", + "Mohammad Shahed Sorower" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.05467", + "source": "arxiv", + "source_id": "arxiv:2601.05467", + "pdf_url": "https://arxiv.org/pdf/2601.05467", + "primary_query": "function-calling" + }, + { + "id": "2510.26167", + "title": "ToolRM: Towards Agentic Tool-Use Reward Modeling", + "url": "https://arxiv.org/abs/2510.26167", + "published": "2025-10-30", + "updated": "2026-01-13", + "authors": [ + "Renhao Li", + "Jianhong Tu", + "Yang Su", + "Yantao Liu", + "Fei Huang", + "Hamid Alinejad-Rokny", + "Derek F. Wong", + "Junyang Lin", + "Min Yang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.26167", + "source": "arxiv", + "source_id": "arxiv:2510.26167", + "pdf_url": "https://arxiv.org/pdf/2510.26167", + "primary_query": "function-calling" + }, + { + "id": "2510.22768", + "title": "Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion", + "url": "https://arxiv.org/abs/2510.22768", + "published": "2025-10-26", + "updated": "2026-06-02", + "authors": [ + "Haoyi Qiu", + "Yilun Zhou", + "Pranav Narayanan Venkit", + "Kung-Hsiang Huang", + "Jiaxin Zhang", + "Nanyun Peng", + "Chien-Sheng Wu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.22768", + "source": "arxiv", + "source_id": "arxiv:2510.22768", + "pdf_url": "https://arxiv.org/pdf/2510.22768", + "primary_query": "function-calling" + }, + { + "id": "2607.05794", + "title": "From Passive Retrieval to Active Memory Navigation: Learning to Use Memory as a Structured Action Space", + "url": "https://arxiv.org/abs/2607.05794", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Yue Xu", + "Yutao Sun", + "Yihao Liu", + "Mengyu Zhou", + "Jiayi Qiao", + "Lu Ma", + "Kai Tang", + "Wenjie Wang", + "Xiaoxi Jiang", + "Guanjun Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.05794", + "source": "arxiv", + "source_id": "arxiv:2607.05794", + "pdf_url": "https://arxiv.org/pdf/2607.05794", + "primary_query": "tool-use" + }, + { + "id": "2607.06341", + "title": "Harnessing Code Agents for Automatic Software Verification", + "url": "https://arxiv.org/abs/2607.06341", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Shuangxiang Kan", + "Shuanglong Kan", + "Sebastian Ertel" + ], + "categories": [ + "cs.FL", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.06341", + "source": "arxiv", + "source_id": "arxiv:2607.06341", + "pdf_url": "https://arxiv.org/pdf/2607.06341", + "primary_query": "coding-agent" + }, + { + "id": "2607.05001", + "title": "TACTIC-KG: Toward Small Agent Teams for Cyber Threat Intelligence Knowledge Graph Construction", + "url": "https://arxiv.org/abs/2607.05001", + "published": "2026-07-06", + "updated": "2026-07-07", + "authors": [ + "Mouhamed Amine Bouchiha", + "Gregory Blanc" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.05001", + "source": "arxiv", + "source_id": "arxiv:2607.05001", + "pdf_url": "https://arxiv.org/pdf/2607.05001", + "primary_query": "llm-agent" + }, + { + "id": "2607.05666", + "title": "What Do AI Agents Actually Change? An Empirical Taxonomy of Mutation Patterns in Performance-Improving Pull Requests", + "url": "https://arxiv.org/abs/2607.05666", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Illia Dovhoshliubnyi", + "Nima Soroush", + "Ashkan Sami", + "Alexander Brownlee" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2607.05666", + "source": "arxiv", + "source_id": "arxiv:2607.05666", + "pdf_url": "https://arxiv.org/pdf/2607.05666", + "primary_query": "ai-agent" + }, + { + "id": "2607.05518", + "title": "aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents", + "url": "https://arxiv.org/abs/2607.05518", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Sai Varun Kodathala" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.05518", + "source": "arxiv", + "source_id": "arxiv:2607.05518", + "pdf_url": "https://arxiv.org/pdf/2607.05518", + "primary_query": "ai-agent" + }, + { + "id": "2607.04697", + "title": "AI Agent Pull Requests on GitHub: Frequency, Structure, and Merge Conflict Rates", + "url": "https://arxiv.org/abs/2607.04697", + "published": "2026-07-06", + "updated": "2026-07-07", + "authors": [ + "George Xu", + "Arjun Subramanian", + "Nithilan Karthik" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2607.04697", + "source": "arxiv", + "source_id": "arxiv:2607.04697", + "pdf_url": "https://arxiv.org/pdf/2607.04697", + "primary_query": "ai-agent" + }, + { + "id": "2607.05391", + "title": "LLM-as-a-Verifier: A General-Purpose Verification Framework", + "url": "https://arxiv.org/abs/2607.05391", + "published": "2026-07-06", + "updated": "2026-07-07", + "authors": [ + "Jacky Kwok", + "Shulu Li", + "Pranav Atreya", + "Yuejiang Liu", + "Yixing Jiang", + "Chelsea Finn", + "Marco Pavone", + "Ion Stoica", + "Azalia Mirhoseini" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.05391", + "source": "arxiv", + "source_id": "arxiv:2607.05391", + "pdf_url": "https://arxiv.org/pdf/2607.05391", + "primary_query": "coding-agent" + }, + { + "id": "2607.04334", + "title": "Do GUI Agents Believe Their Eyes? Diagnosing State-Belief Reliance on Pixels versus Structure", + "url": "https://arxiv.org/abs/2607.04334", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Guijia Zhang", + "Harry Yang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.04334", + "source": "arxiv", + "source_id": "arxiv:2607.04334", + "pdf_url": "https://arxiv.org/pdf/2607.04334", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.04394", + "title": "MechMath Agent Team: LLM Driven Agents for Mathematical Research", + "url": "https://arxiv.org/abs/2607.04394", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Yichuan Cao", + "Ruichen Qiu", + "Junqi Liu", + "Jiaqi Wang", + "Dakai Guo", + "Ruyong Feng", + "Lihong Zhi", + "Xiao-Shan Gao" + ], + "categories": [ + "cs.AI", + "cs.SC" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.04394", + "source": "arxiv", + "source_id": "arxiv:2607.04394", + "pdf_url": "https://arxiv.org/pdf/2607.04394", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.02942", + "title": "A Workflow-Aware Serving Layer for Agentic Applications", + "url": "https://arxiv.org/abs/2607.02942", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Jiayi Qian", + "Zishen Wan", + "Hanchen Yang", + "Chun Tao", + "Souvik Kundu", + "Tushar Krishna" + ], + "categories": [ + "cs.DC", + "cs.MA" + ], + "topics": [ + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02942", + "source": "arxiv", + "source_id": "arxiv:2607.02942", + "pdf_url": "https://arxiv.org/pdf/2607.02942", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03105", + "title": "ORBIT-Q: Dual-axis benchmarking of autonomous agents in scientific quantum programming", + "url": "https://arxiv.org/abs/2607.03105", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Shi-Xin Zhang", + "Yu-Qin Chen" + ], + "categories": [ + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03105", + "source": "arxiv", + "source_id": "arxiv:2607.03105", + "pdf_url": "https://arxiv.org/pdf/2607.03105", + "primary_query": "coding-agent" + }, + { + "id": "2607.03316", + "title": "Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild", + "url": "https://arxiv.org/abs/2607.03316", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Hong Yi Lin", + "Mingzhao Liang", + "Kla Tantithamthavorn", + "Patanamon Thongtanunam" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2607.03316", + "source": "arxiv", + "source_id": "arxiv:2607.03316", + "pdf_url": "https://arxiv.org/pdf/2607.03316", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.03220", + "title": "CONTRA: Red-Teaming Configurations of Personalizable Agents", + "url": "https://arxiv.org/abs/2607.03220", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Jonathan Nöther", + "Adish Singla", + "Goran Radanovic" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2607.03220", + "source": "arxiv", + "source_id": "arxiv:2607.03220", + "pdf_url": "https://arxiv.org/pdf/2607.03220", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.01812", + "title": "TO-Master: an LLM-agent framework for automated topology optimization", + "url": "https://arxiv.org/abs/2607.01812", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Haoju Lin", + "Wenchang Zhang", + "Weipeng Xu", + "Xiang Li", + "Tian Xu", + "Tianju Xue" + ], + "categories": [ + "cs.CE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01812", + "source": "arxiv", + "source_id": "arxiv:2607.01812", + "pdf_url": "https://arxiv.org/pdf/2607.01812", + "primary_query": "llm-agent" + }, + { + "id": "2607.02453", + "title": "Adoption and Ecosystem Health: A Longitudinal Analysis of Open-Source Multi-Agent Frameworks", + "url": "https://arxiv.org/abs/2607.02453", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Xi Zhang", + "Papi Menon", + "Vivian Chu", + "Koray Cosguner" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.02453", + "source": "arxiv", + "source_id": "arxiv:2607.02453", + "pdf_url": "https://arxiv.org/pdf/2607.02453", + "primary_query": "ai-agent" + }, + { + "id": "2607.02245", + "title": "Copewell: A Multi-Agent Swarm Architecture for Equitable Mental Wellness Support", + "url": "https://arxiv.org/abs/2607.02245", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Seren Yenikent", + "Jack Vinijtrongjit", + "Katherine Ng" + ], + "categories": [ + "cs.AI", + "cs.CY", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.02245", + "source": "arxiv", + "source_id": "arxiv:2607.02245", + "pdf_url": "https://arxiv.org/pdf/2607.02245", + "primary_query": "ai-agent" + }, + { + "id": "2607.02381", + "title": "HULAT2 at MER-TRANS 2026: Governed Multi-Agent Simplification for Spanish Easy-to-Read Generation", + "url": "https://arxiv.org/abs/2607.02381", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Lourdes Moreno", + "Paloma Martínez", + "Marco Antonio Sanchez-Escudero", + "Miguel Domínguez-Gómez" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02381", + "source": "arxiv", + "source_id": "arxiv:2607.02381", + "pdf_url": "https://arxiv.org/pdf/2607.02381", + "primary_query": "agentic-ai" + }, + { + "id": "2607.01531", + "title": "OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration", + "url": "https://arxiv.org/abs/2607.01531", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "David Courtis", + "Wenhao Li", + "Scott Sanner" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.01531", + "source": "arxiv", + "source_id": "arxiv:2607.01531", + "pdf_url": "https://arxiv.org/pdf/2607.01531", + "primary_query": "llm-agent" + }, + { + "id": "2607.01047", + "title": "Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates", + "url": "https://arxiv.org/abs/2607.01047", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Elias Najarro", + "Ane Espeseth", + "Eleni Nisioti", + "Sebastian Risi", + "Stefano Nichele" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "memory", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01047", + "source": "arxiv", + "source_id": "arxiv:2607.01047", + "pdf_url": "https://arxiv.org/pdf/2607.01047", + "primary_query": "llm-agent" + }, + { + "id": "2607.01510", + "title": "Janus: a Playground for User-Involved Agentic Permission Management", + "url": "https://arxiv.org/abs/2607.01510", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Natalie Grace Brigham", + "Eugene Bagdasarian", + "Tadayoshi Kohno", + "Franziska Roesner" + ], + "categories": [ + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.01510", + "source": "arxiv", + "source_id": "arxiv:2607.01510", + "pdf_url": "https://arxiv.org/pdf/2607.01510", + "primary_query": "ai-agent" + }, + { + "id": "2607.00407", + "title": "Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising", + "url": "https://arxiv.org/abs/2607.00407", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Tianci Liu", + "Zihan Dong", + "Linjun Zhang", + "Haoyu Wang", + "jing Gao", + "Emre Kiciman", + "Ranveer Chandra", + "Wei-Ting Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "multi-agent", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.00407", + "source": "arxiv", + "source_id": "arxiv:2607.00407", + "pdf_url": "https://arxiv.org/pdf/2607.00407", + "primary_query": "ai-agent" + }, + { + "id": "2607.01366", + "title": "Auto-FL-Research: Agentic Search for Federated Learning Algorithms", + "url": "https://arxiv.org/abs/2607.01366", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Holger R. Roth", + "Ziyue Xu", + "Chester Chen", + "Daguang Xu", + "Peter Cnudde", + "Andrew Feng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2607.01366", + "source": "arxiv", + "source_id": "arxiv:2607.01366", + "pdf_url": "https://arxiv.org/pdf/2607.01366", + "primary_query": "agentic-ai" + }, + { + "id": "2607.00990", + "title": "SWE-Doctor: Guiding Software Engineering Agents with Runtime Diagnosis from Multi-Faceted Bug Reproduction Tests", + "url": "https://arxiv.org/abs/2607.00990", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Yaoqi Guo", + "Yang Liu", + "Jie M. Zhang", + "Yun Ma", + "Yiling Lou", + "Zhenpeng Chen" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.00990", + "source": "arxiv", + "source_id": "arxiv:2607.00990", + "pdf_url": "https://arxiv.org/pdf/2607.00990", + "primary_query": "coding-agent" + }, + { + "id": "2606.31471", + "title": "Think While You Map: Asynchronous Vision-Language Agents for Incremental 3D Scene Graphs", + "url": "https://arxiv.org/abs/2606.31471", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Deniz Bickici", + "Michael Pabst", + "Shohei Mori", + "Dieter Schmalstieg" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.31471", + "source": "arxiv", + "source_id": "arxiv:2606.31471", + "pdf_url": "https://arxiv.org/pdf/2606.31471", + "primary_query": "language-agent" + }, + { + "id": "2606.31831", + "title": "An Agentic AI Framework to Accelerate Scientific Discovery in Plant Phenotyping", + "url": "https://arxiv.org/abs/2606.31831", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Renan Souza", + "Daniel Rosendo", + "Kelsey Carter", + "John Lagergren", + "Frédéric Suter", + "Shelaine L. Curd", + "Gerald A. Tuskan", + "Rafael Ferreira da Silva", + "David Weston" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent" + ], + "arxiv_id": "2606.31831", + "source": "arxiv", + "source_id": "arxiv:2606.31831", + "pdf_url": "https://arxiv.org/pdf/2606.31831", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31916", + "title": "Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action", + "url": "https://arxiv.org/abs/2606.31916", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Ben Slater", + "Matteo G. Mecattaf", + "Lucy G. Cheke", + "John Burden", + "Winnie Street" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.31916", + "source": "arxiv", + "source_id": "arxiv:2606.31916", + "pdf_url": "https://arxiv.org/pdf/2606.31916", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.31767", + "title": "JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java", + "url": "https://arxiv.org/abs/2606.31767", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Khashayar Etemadi", + "Zhendong Su" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.31767", + "source": "arxiv", + "source_id": "arxiv:2606.31767", + "pdf_url": "https://arxiv.org/pdf/2606.31767", + "primary_query": "coding-agent" + }, + { + "id": "2606.31665", + "title": "ForecastAgentSearch: Towards a Multi-Expert Agent Search System for Geopolitical Event Forecasting", + "url": "https://arxiv.org/abs/2606.31665", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Miaomiao Cai", + "He Chang", + "Yunshan Ma", + "See-kiong Ng" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31665", + "source": "arxiv", + "source_id": "arxiv:2606.31665", + "pdf_url": "https://arxiv.org/pdf/2606.31665", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29719", + "title": "A Diagnostic Framework and Multi-Evaluator Audit of Evaluator-Driven Preference Dynamics in Self-Adapting LLM Agents", + "url": "https://arxiv.org/abs/2606.29719", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Liu Zewen" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29719", + "source": "arxiv", + "source_id": "arxiv:2606.29719", + "pdf_url": "https://arxiv.org/pdf/2606.29719", + "primary_query": "llm-agent" + }, + { + "id": "2606.29745", + "title": "ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit", + "url": "https://arxiv.org/abs/2606.29745", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Abhijnan Nath", + "Nikhil Krishnaswamy" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.29745", + "source": "arxiv", + "source_id": "arxiv:2606.29745", + "pdf_url": "https://arxiv.org/pdf/2606.29745", + "primary_query": "language-agent" + }, + { + "id": "2606.30970", + "title": "Behavioral Governance for Autonomous AI Agents: The AgentBound Framework", + "url": "https://arxiv.org/abs/2606.30970", + "published": "2026-06-29", + "updated": "2026-07-01", + "authors": [ + "Anuj Kaul", + "Qianlong Lan", + "Pranay Gupta" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.30970", + "source": "arxiv", + "source_id": "arxiv:2606.30970", + "pdf_url": "https://arxiv.org/pdf/2606.30970", + "primary_query": "ai-agent" + }, + { + "id": "2606.29788", + "title": "MemLeak: Diagnosing Information Leaks in Multimodal Agent Memory", + "url": "https://arxiv.org/abs/2606.29788", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Kuan Wang", + "Chao Zhang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "ai-agent" + ], + "arxiv_id": "2606.29788", + "source": "arxiv", + "source_id": "arxiv:2606.29788", + "pdf_url": "https://arxiv.org/pdf/2606.29788", + "primary_query": "agent-memory" + }, + { + "id": "2606.30616", + "title": "Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent", + "url": "https://arxiv.org/abs/2606.30616", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Lei Bai", + "Zongsheng Cao", + "Yang Chen", + "Zhiyao Cui", + "Shangheng Du", + "Yue Fan", + "Shiyang Feng", + "Zijie Guo", + "Haonan He", + "Liang He", + "Xiaohan He", + "Shuyue Hu", + "Yusong Hu", + "Songtao Huang", + "Yichen Jiang", + "Hao Li", + "Xin Li", + "Dahua Lin", + "Weihao Lin", + "Fenghua Ling", + "Dongrui Liu", + "Zhuo Liu", + "Runmin Ma", + "Chunjiang Mu", + "Haoyang Peng", + "Tianshuo Peng", + "Jinxin Shi", + "Luohe Shi", + "Boyuan Sun", + "Zelin Tan", + "Shengji Tang", + "Qianyi Wang", + "Yiming Wu", + "Yi Xie", + "Xiangchao Yan", + "Jingqi Ye", + "Peng Ye", + "Fangchen Yu", + "Jiakang Yuan", + "Bihao Zhan", + "Bo Zhang", + "Chen Zhang", + "Shufei Zhang", + "Shuaiyu Zhang", + "Wenlong Zhang", + "Yiqun Zhang", + "Junpeng Zhao", + "Zhijie Zhong", + "Bowen Zhou", + "Yuhao Zhou" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.30616", + "source": "arxiv", + "source_id": "arxiv:2606.30616", + "pdf_url": "https://arxiv.org/pdf/2606.30616", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.30111", + "title": "Automating the Design of Embodied Agent Architectures", + "url": "https://arxiv.org/abs/2606.30111", + "published": "2026-06-29", + "updated": "2026-07-03", + "authors": [ + "Jian Zhou", + "Sihao Lin", + "Jin Li", + "Shuai Fu", + "Gengze Zhou", + "Qi Wu" + ], + "categories": [ + "cs.RO", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.30111", + "source": "arxiv", + "source_id": "arxiv:2606.30111", + "pdf_url": "https://arxiv.org/pdf/2606.30111", + "primary_query": "coding-agent" + }, + { + "id": "2606.29495", + "title": "Cognitive World Models for Process-Level Social Influence Evaluation", + "url": "https://arxiv.org/abs/2606.29495", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Minghui Ma", + "Bin Guo", + "Han Wang", + "Mengqi Chen", + "Jingqi Liu", + "Yan Liu", + "Zhiwen Yu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.29495", + "source": "arxiv", + "source_id": "arxiv:2606.29495", + "pdf_url": "https://arxiv.org/pdf/2606.29495", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28896", + "title": "A Task-Driven and Quality-Assured Agent Framework for SAR Data Generation", + "url": "https://arxiv.org/abs/2606.28896", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Xuanting Wu", + "Fan Zhanga", + "Fei Ma", + "Ling Guan", + "Guochun Ma", + "Yongsheng Zhou" + ], + "categories": [ + "eess.IV", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.28896", + "source": "arxiv", + "source_id": "arxiv:2606.28896", + "pdf_url": "https://arxiv.org/pdf/2606.28896", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.28279", + "title": "Agentic Hardware Design as Repository-Level Code Evolution", + "url": "https://arxiv.org/abs/2606.28279", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Cunxi Yu", + "Chenhui Deng", + "Nathaniel Pinckney", + "Brucek Khailany" + ], + "categories": [ + "cs.AR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.28279", + "source": "arxiv", + "source_id": "arxiv:2606.28279", + "pdf_url": "https://arxiv.org/pdf/2606.28279", + "primary_query": "agentic-ai" + }, + { + "id": "2606.28430", + "title": "Building to the Test: Coding Agents Deliver What You Check, Not What You Requested", + "url": "https://arxiv.org/abs/2606.28430", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Yanuo Ma", + "Ben Kereopa-Yorke", + "Ben Schultz" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.28430", + "source": "arxiv", + "source_id": "arxiv:2606.28430", + "pdf_url": "https://arxiv.org/pdf/2606.28430", + "primary_query": "coding-agent" + }, + { + "id": "2606.28187", + "title": "GBC: Gradient-Based Connections for Optimizing Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.28187", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Xiaocheng Yang", + "Abdulrahman Alrabah", + "Dilek Hakkani-Tür", + "Gokhan Tur" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "multi-agent", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28187", + "source": "arxiv", + "source_id": "arxiv:2606.28187", + "pdf_url": "https://arxiv.org/pdf/2606.28187", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.27416", + "title": "Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents", + "url": "https://arxiv.org/abs/2606.27416", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Vassili Philippov", + "Pavel Katunin", + "Dmitry Andreev", + "Igor Ostanin", + "Anton Nikolaev" + ], + "categories": [ + "cs.MA", + "cs.SE" + ], + "topics": [ + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.27416", + "source": "arxiv", + "source_id": "arxiv:2606.27416", + "pdf_url": "https://arxiv.org/pdf/2606.27416", + "primary_query": "coding-agent" + }, + { + "id": "2606.26649", + "title": "Autoformalization of Agent Instructions into Policy-as-Code", + "url": "https://arxiv.org/abs/2606.26649", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Adam Mondl", + "Matthew Maisel", + "John H. Brock" + ], + "categories": [ + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.26649", + "source": "arxiv", + "source_id": "arxiv:2606.26649", + "pdf_url": "https://arxiv.org/pdf/2606.26649", + "primary_query": "agent-safety" + }, + { + "id": "2606.26057", + "title": "The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems", + "url": "https://arxiv.org/abs/2606.26057", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Seth Dobrin", + "Łukasz Chmiel" + ], + "categories": [ + "cs.AI", + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.26057", + "source": "arxiv", + "source_id": "arxiv:2606.26057", + "pdf_url": "https://arxiv.org/pdf/2606.26057", + "primary_query": "ai-agent" + }, + { + "id": "2606.26356", + "title": "Instruction Bleed: Cross-Module Interference in Prompt-Composed Agentic Systems", + "url": "https://arxiv.org/abs/2606.26356", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Ching-Yu Lin", + "Yifan Liu" + ], + "categories": [ + "cs.AI", + "cs.IR", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "agentic-ai" + ], + "arxiv_id": "2606.26356", + "source": "arxiv", + "source_id": "arxiv:2606.26356", + "pdf_url": "https://arxiv.org/pdf/2606.26356", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.25519", + "title": "Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models", + "url": "https://arxiv.org/abs/2606.25519", + "published": "2026-06-24", + "updated": "2026-06-29", + "authors": [ + "Xinyu Lian", + "Walid Krichene", + "Beichen Huang", + "Masahiro Tanaka", + "Olatunji Ruwase", + "Li Zhang", + "Minjia Zhang" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.25519", + "source": "arxiv", + "source_id": "arxiv:2606.25519", + "pdf_url": "https://arxiv.org/pdf/2606.25519", + "primary_query": "tool-use" + }, + { + "id": "2606.25195", + "title": "SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward", + "url": "https://arxiv.org/abs/2606.25195", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Rupam Patir", + "Keyan Guo", + "Haipeng Cai", + "Hongxin Hu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2606.25195", + "source": "arxiv", + "source_id": "arxiv:2606.25195", + "pdf_url": "https://arxiv.org/pdf/2606.25195", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24416", + "title": "Agentic AI for Bilevel Long-Term Optimization of Policy-Driven Physical Layer Systems", + "url": "https://arxiv.org/abs/2606.24416", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Bingnan Xiao", + "Chenhao Yang", + "Wei Ni", + "Xin Wang", + "Tony Q. S. Quek" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.24416", + "source": "arxiv", + "source_id": "arxiv:2606.24416", + "pdf_url": "https://arxiv.org/pdf/2606.24416", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24855", + "title": "OpenThoughts-Agent: Data Recipes for Agentic Models", + "url": "https://arxiv.org/abs/2606.24855", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Negin Raoof", + "Richard Zhuang", + "Marianna Nezhurina", + "Etash Guha", + "Atula Tejaswi", + "Ryan Marten", + "Charlie F. Ruan", + "Tyler Griggs", + "Alexander Glenn Shaw", + "Hritik Bansal", + "E. Kelly Buchanan", + "Artem Gazizov", + "Reinhard Heckel", + "Chinmay Hegde", + "Sankalp Jajee", + "Daanish Khazi", + "Emmanouil Koukoumidis", + "Xiangyi Li", + "Hange Liu", + "Shlok Natarajan", + "Harsh Raj", + "Nicholas Roberts", + "Ethan Shen", + "Nishad Singhi", + "Michael Siu", + "Ashima Suvarna", + "Hanwen Xing", + "Patrick Yubeaton", + "Robert Zhang", + "Leon Liangyu Chen", + "Xiaokun Chen", + "Steven Dillmann", + "Saadia Gabriel", + "Xunyi Jiang", + "Anurag Kashyap", + "Boxuan Li", + "Yein Park", + "Minh Pham", + "Sujay Sanghavi", + "Lin Shi", + "Ke Sun", + "Yixin Wang", + "Zhiwei Xu", + "Erica Zhang", + "Siyan Zhao", + "Wanjia Zhao", + "Jenia Jitsev", + "Alex Dimakis", + "Benjamin Feuer", + "Ludwig Schmidt" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.24855", + "source": "arxiv", + "source_id": "arxiv:2606.24855", + "pdf_url": "https://arxiv.org/pdf/2606.24855", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.25191", + "title": "To Isolate or to Score? Model-Adaptive Assessment for Cost-Efficient Multi-Agent RAG", + "url": "https://arxiv.org/abs/2606.25191", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Jungseob Lee", + "Chanjun Park", + "Heuiseok Lim" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.25191", + "source": "arxiv", + "source_id": "arxiv:2606.25191", + "pdf_url": "https://arxiv.org/pdf/2606.25191", + "primary_query": "rag-agent" + }, + { + "id": "2606.23130", + "title": "Understanding the (In)Security of Vibe-Coded Applications", + "url": "https://arxiv.org/abs/2606.23130", + "published": "2026-06-22", + "updated": "2026-06-23", + "authors": [ + "Junquan Deng", + "Zhiyu Fan", + "Ruijie Meng" + ], + "categories": [ + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.23130", + "source": "arxiv", + "source_id": "arxiv:2606.23130", + "pdf_url": "https://arxiv.org/pdf/2606.23130", + "primary_query": "ai-agent" + }, + { + "id": "2606.22953", + "title": "Plans Don't Persist: Why Context Management Is Load Bearing for LLM Agents", + "url": "https://arxiv.org/abs/2606.22953", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Aman Mehta", + "Anupam Datta" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.22953", + "source": "arxiv", + "source_id": "arxiv:2606.22953", + "pdf_url": "https://arxiv.org/pdf/2606.22953", + "primary_query": "planning-agent" + }, + { + "id": "2606.21836", + "title": "AgentDSE: Reasoning-Augmented Architectural Design Space Exploration", + "url": "https://arxiv.org/abs/2606.21836", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Chenyu Wang", + "Jiahe Caroline Shi", + "David Kong", + "Duane S. Boning", + "Zishen Wan", + "Yilun Du", + "Vijay Janapa Reddi" + ], + "categories": [ + "cs.AR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.21836", + "source": "arxiv", + "source_id": "arxiv:2606.21836", + "pdf_url": "https://arxiv.org/pdf/2606.21836", + "primary_query": "coding-agent" + }, + { + "id": "2606.21401", + "title": "SwarmX: Agentic Scheduling for Low-Latency Agentic Systems", + "url": "https://arxiv.org/abs/2606.21401", + "published": "2026-06-19", + "updated": "2026-06-28", + "authors": [ + "Yeqi Huang", + "Yanwei Ye", + "Guomin Chen", + "Wenhao Su", + "Bin Gong", + "Jialian Li", + "Zhan Lu", + "Yangshen Deng", + "Xuan Sun", + "Le Xu", + "Luo Mai" + ], + "categories": [ + "cs.DC", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.21401", + "source": "arxiv", + "source_id": "arxiv:2606.21401", + "pdf_url": "https://arxiv.org/pdf/2606.21401", + "primary_query": "agentic-ai" + }, + { + "id": "2606.21228", + "title": "Sakana Fugu Technical Report", + "url": "https://arxiv.org/abs/2606.21228", + "published": "2026-06-19", + "updated": "2026-06-23", + "authors": [ + "Yujin Tang", + "Edoardo Cetin", + "Jinglue Xu", + "Qi Sun", + "Stefan Nielsen", + "Vincent Richard", + "Haruto Goda", + "Iaroslav Tymchenko", + "Nhan Nguyen", + "Hyunin Lee", + "Mari Ashiga", + "Shashank Kotyan", + "So Kuroki", + "Tarin Clanuwat" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "coding-agent", + "multi-agent", + "rag", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.21228", + "source": "arxiv", + "source_id": "arxiv:2606.21228", + "pdf_url": "https://arxiv.org/pdf/2606.21228", + "primary_query": "coding-agent" + }, + { + "id": "2606.20510", + "title": "Efficient and Sound Probabilistic Verification for AI Agents", + "url": "https://arxiv.org/abs/2606.20510", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Alaia Solko-Breslin", + "Pramod Kaushik Mudrakarta", + "Mihai Christodorescu", + "Somesh Jha", + "Krishnamurthy Dj Dvijotham" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.20510", + "source": "arxiv", + "source_id": "arxiv:2606.20510", + "pdf_url": "https://arxiv.org/pdf/2606.20510", + "primary_query": "ai-agent" + }, + { + "id": "2606.19242", + "title": "Runtime Compliance Verification for AI Agents", + "url": "https://arxiv.org/abs/2606.19242", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Nafiseh Kahani", + "Masoud Barati", + "Diana Addae" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "function-calling", + "tool-use" + ], + "arxiv_id": "2606.19242", + "source": "arxiv", + "source_id": "arxiv:2606.19242", + "pdf_url": "https://arxiv.org/pdf/2606.19242", + "primary_query": "ai-agent" + }, + { + "id": "2606.28374", + "title": "Recursive Self-Evolving Agents via Held-Out Selection", + "url": "https://arxiv.org/abs/2606.28374", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Michael Nguyen", + "Quoc Nguyen", + "Paul Vuong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.28374", + "source": "arxiv", + "source_id": "arxiv:2606.28374", + "pdf_url": "https://arxiv.org/pdf/2606.28374", + "primary_query": "tool-use" + }, + { + "id": "2606.18363", + "title": "Guava: An Effective and Universal Harness for Embodied Manipulation", + "url": "https://arxiv.org/abs/2606.18363", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Haowen Liu", + "Xirui Li", + "Shaoxiong Yao", + "Peng Shi", + "Tianyi Zhou", + "Jia-Bin Huang", + "Furong Huang", + "Jiayuan Mao" + ], + "categories": [ + "cs.RO", + "cs.AI" + ], + "topics": [ + "embodied-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "tool-use" + ], + "arxiv_id": "2606.18363", + "source": "arxiv", + "source_id": "arxiv:2606.18363", + "pdf_url": "https://arxiv.org/pdf/2606.18363", + "primary_query": "agentic-ai" + }, + { + "id": "2606.17453", + "title": "MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors", + "url": "https://arxiv.org/abs/2606.17453", + "published": "2026-06-16", + "updated": "2026-06-17", + "authors": [ + "Lubin Bai", + "Mengyu Cao", + "Sixue Wang", + "Zhongwei Wan", + "Yue Pan", + "Jiale Hou", + "Xiang Li", + "Xiuyuan Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.17453", + "source": "arxiv", + "source_id": "arxiv:2606.17453", + "pdf_url": "https://arxiv.org/pdf/2606.17453", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18023", + "title": "LoopCoder-v2: Only Loop Once for Efficient Test-Time Computation Scaling", + "url": "https://arxiv.org/abs/2606.18023", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Jian Yang", + "Shawn Guo", + "Wei Zhang", + "Tianyu Zheng", + "Yaxin Du", + "Haau-Sing Li", + "Jiajun Wu", + "Yue Song", + "Yan Xing", + "Qingsong Cai", + "Zelong Huang", + "Chuan Hao", + "Ran Tao", + "Xianglong Liu", + "Wayne Xin Zhao", + "Mingjie Tang", + "Weifeng Lv", + "Ming Zhou", + "Bryan Dai" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.18023", + "source": "arxiv", + "source_id": "arxiv:2606.18023", + "pdf_url": "https://arxiv.org/pdf/2606.18023", + "primary_query": "tool-use" + }, + { + "id": "2606.17383", + "title": "Model Validation of Agentic AI Systems: A POMDP-Based Framework for Belief-State, Forecast, and Policy Validation", + "url": "https://arxiv.org/abs/2606.17383", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Matthew Francis Dixon" + ], + "categories": [ + "q-fin.RM", + "cs.AI", + "cs.LG", + "stat.ML" + ], + "topics": [ + "agent-safety", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.17383", + "source": "arxiv", + "source_id": "arxiv:2606.17383", + "pdf_url": "https://arxiv.org/pdf/2606.17383", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.16813", + "title": "GIST-CMTF: Goal-State Inference for Causal Minimal Tool Filtering in LLM Agents", + "url": "https://arxiv.org/abs/2606.16813", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Rahul Suresh Babu", + "Rohit Shukla" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.16813", + "source": "arxiv", + "source_id": "arxiv:2606.16813", + "pdf_url": "https://arxiv.org/pdf/2606.16813", + "primary_query": "tool-use" + }, + { + "id": "2606.16839", + "title": "Towards LLM Accelerated Rapid Reviews for Software Tool Discovery -- Case for Log Anomaly Detection", + "url": "https://arxiv.org/abs/2606.16839", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Jesse Nyyssölä", + "Hamza Bin Mazhar", + "Alexander Bakhtin", + "Matteo Esposito", + "Nana Reinikainen", + "Yuqing Wang", + "Ying Song", + "Davide Taibi", + "Mika Mäntylä" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.16839", + "source": "arxiv", + "source_id": "arxiv:2606.16839", + "pdf_url": "https://arxiv.org/pdf/2606.16839", + "primary_query": "planning-agent" + }, + { + "id": "2606.16481", + "title": "Steering Emotional Dynamics for Art Therapy: Controllable Narrative Script Generation through Hierarchically Guided LLM Agents", + "url": "https://arxiv.org/abs/2606.16481", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Suqing Wang", + "Qinghai Miao", + "Chao Guo", + "Yisheng Lv" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.16481", + "source": "arxiv", + "source_id": "arxiv:2606.16481", + "pdf_url": "https://arxiv.org/pdf/2606.16481", + "primary_query": "planning-agent" + }, + { + "id": "2606.17368", + "title": "Distributed General-Purpose Agent Networks: Architecture, Key Mechanisms, and Prototypes", + "url": "https://arxiv.org/abs/2606.17368", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Shengli Zhang", + "Deen Ma", + "Zibin Lin", + "Taotao Wang" + ], + "categories": [ + "cs.AI", + "cs.NI" + ], + "topics": [ + "computer-use", + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.17368", + "source": "arxiv", + "source_id": "arxiv:2606.17368", + "pdf_url": "https://arxiv.org/pdf/2606.17368", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.15709", + "title": "AI-Driven Framework for Adaptive Water Network Management with Proof-of-Concept Implementation: Addressing Non-Revenue Water in Jordan", + "url": "https://arxiv.org/abs/2606.15709", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Mohammed Fasha", + "Nahel Al-Maayta", + "Bilal Sowan", + "Mohammad Athamneh", + "Husam Barham" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling", + "rag-agent" + ], + "arxiv_id": "2606.15709", + "source": "arxiv", + "source_id": "arxiv:2606.15709", + "pdf_url": "https://arxiv.org/pdf/2606.15709", + "primary_query": "function-calling" + }, + { + "id": "2606.15874", + "title": "LLM-as-Code: Agentic Programming for Agent Harness", + "url": "https://arxiv.org/abs/2606.15874", + "published": "2026-06-14", + "updated": "2026-06-22", + "authors": [ + "Junjia Qi", + "Zichuan Fu", + "Jingtong Gao", + "Wenlin Zhang", + "Hanyu Yan", + "Xian Wu", + "Xiangyu Zhao" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.15874", + "source": "arxiv", + "source_id": "arxiv:2606.15874", + "pdf_url": "https://arxiv.org/pdf/2606.15874", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.15906", + "title": "MAGE-RAG: Multigranular Adaptive Graph Evidence for Agentic Multimodal RAG in Long-Document QA", + "url": "https://arxiv.org/abs/2606.15906", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Yilong Zuo", + "Xunkai Li", + "Jing Yuan", + "Qiangqiang Dai", + "Hongchao Qin", + "Ronghua Li" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.CL", + "cs.DB", + "cs.MM" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.15906", + "source": "arxiv", + "source_id": "arxiv:2606.15906", + "pdf_url": "https://arxiv.org/pdf/2606.15906", + "primary_query": "rag-agent" + }, + { + "id": "2606.15994", + "title": "Agentic Framework for Deep Learning workload migration via In-Context Learning", + "url": "https://arxiv.org/abs/2606.15994", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Qiyue Liang", + "Steven Ingram", + "George Vanica", + "Andi Gavrilescu", + "Newfel Harrat", + "Hassan Sipra", + "Sethuraman Sankaran" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-safety", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.15994", + "source": "arxiv", + "source_id": "arxiv:2606.15994", + "pdf_url": "https://arxiv.org/pdf/2606.15994", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.15034", + "title": "OSGuard: A Benchmark for Safety in Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.15034", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Mina Mohammadmirzaei", + "Jeffrey Flanigan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.15034", + "source": "arxiv", + "source_id": "arxiv:2606.15034", + "pdf_url": "https://arxiv.org/pdf/2606.15034", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.12837", + "title": "LoHoSearch: Benchmarking Long-Horizon Search Agents Beyond the Human Difficulty Ceiling", + "url": "https://arxiv.org/abs/2606.12837", + "published": "2026-06-11", + "updated": "2026-06-17", + "authors": [ + "Jiarui Zhao", + "Rongzhi Zhang", + "Lingchuan Liu", + "Hao Yang", + "Xunliang Cai", + "Xi Su" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.12837", + "source": "arxiv", + "source_id": "arxiv:2606.12837", + "pdf_url": "https://arxiv.org/pdf/2606.12837", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.13663", + "title": "HyperTool: Beyond Step-Wise Tool Calls for Tool-Augmented Agents", + "url": "https://arxiv.org/abs/2606.13663", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Yaxin Du", + "Yifan Zhou", + "Yujie Ge", + "Jiajun Wang", + "Xianghe Pang", + "Shuo Tang", + "Tuney Zheng", + "Bryan Dai", + "Jian Yang", + "Siheng Chen" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.13663", + "source": "arxiv", + "source_id": "arxiv:2606.13663", + "pdf_url": "https://arxiv.org/pdf/2606.13663", + "primary_query": "tool-use" + }, + { + "id": "2606.12634", + "title": "Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents", + "url": "https://arxiv.org/abs/2606.12634", + "published": "2026-06-10", + "updated": "2026-06-29", + "authors": [ + "Tianyu Ding", + "Jianhong Xin", + "Juan Pablo De la Cruz Weinstein" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.12634", + "source": "arxiv", + "source_id": "arxiv:2606.12634", + "pdf_url": "https://arxiv.org/pdf/2606.12634", + "primary_query": "tool-use" + }, + { + "id": "2606.11688", + "title": "Goal-Autopilot: A Verifiable Anti-Fabrication Firewall for Unattended Long-Horizon Agents", + "url": "https://arxiv.org/abs/2606.11688", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Youwang Deng" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.11688", + "source": "arxiv", + "source_id": "arxiv:2606.11688", + "pdf_url": "https://arxiv.org/pdf/2606.11688", + "primary_query": "planning-agent" + }, + { + "id": "2606.10394", + "title": "STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios", + "url": "https://arxiv.org/abs/2606.10394", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Sirui Liang", + "Bohan Yu", + "Peiyu Wang", + "Shiguang Guo", + "Wenxing Hu", + "Pengfei Cao", + "Jian Zhao", + "Cao Liu", + "Ke Zeng", + "Xunliang Cai", + "Kang Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.10394", + "source": "arxiv", + "source_id": "arxiv:2606.10394", + "pdf_url": "https://arxiv.org/pdf/2606.10394", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.10532", + "title": "ActiveMem: Distributed Active Memory for Long-Horizon LLM Reasoning", + "url": "https://arxiv.org/abs/2606.10532", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yunhan Jiang", + "Wenbin Duan", + "Shasha Guo", + "Liang Pang", + "Xiaoqian Sun", + "Huawei Shen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.10532", + "source": "arxiv", + "source_id": "arxiv:2606.10532", + "pdf_url": "https://arxiv.org/pdf/2606.10532", + "primary_query": "agent-memory" + }, + { + "id": "2606.11176", + "title": "Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories", + "url": "https://arxiv.org/abs/2606.11176", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Kevin Qinghong Lin", + "Batu EI", + "Yuhong Shi", + "Pan Lu", + "Philip Torr", + "James Zou" + ], + "categories": [ + "cs.CV", + "cs.CL", + "cs.CY", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.11176", + "source": "arxiv", + "source_id": "arxiv:2606.11176", + "pdf_url": "https://arxiv.org/pdf/2606.11176", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.08960", + "title": "Hardening Agent Benchmarks with Adversarial Hacker-Fixer Loops", + "url": "https://arxiv.org/abs/2606.08960", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Ziqian Zhong", + "Ivgeni Segal", + "Ivan Bercovich", + "Shashwat Saxena", + "Kexun Zhang", + "Aditi Raghunathan" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.08960", + "source": "arxiv", + "source_id": "arxiv:2606.08960", + "pdf_url": "https://arxiv.org/pdf/2606.08960", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.09447", + "title": "AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.09447", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Bojie Rong", + "Zheyu Shen", + "Qiaoping Wang", + "Pengfei Kang", + "Yang Xu", + "Yawen Wei", + "Hanyu Wu", + "Zhi Zhao", + "Leihao Pei", + "Linquan Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.09447", + "source": "arxiv", + "source_id": "arxiv:2606.09447", + "pdf_url": "https://arxiv.org/pdf/2606.09447", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09426", + "title": "WeaveBench: A Long-Horizon, Real-World Benchmark for Computer-Use Agents with Hybrid Interfaces", + "url": "https://arxiv.org/abs/2606.09426", + "published": "2026-06-08", + "updated": "2026-07-06", + "authors": [ + "Wanli Li", + "Bowen Zhou", + "Yunyao Yu", + "Zhou Xu", + "Yifan Yang", + "Dongsheng Li", + "Caihua Shan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.09426", + "source": "arxiv", + "source_id": "arxiv:2606.09426", + "pdf_url": "https://arxiv.org/pdf/2606.09426", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09316", + "title": "Anything2Skill: Compiling External Knowledge into Reusable Skills for Agents", + "url": "https://arxiv.org/abs/2606.09316", + "published": "2026-06-08", + "updated": "2026-06-19", + "authors": [ + "Qianjun Pan", + "Yutao Yang", + "Junsong Li", + "Jie Zhou", + "Kai Chen", + "Xin Li", + "Qin Chen", + "Liang He" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.09316", + "source": "arxiv", + "source_id": "arxiv:2606.09316", + "pdf_url": "https://arxiv.org/pdf/2606.09316", + "primary_query": "rag-agent" + }, + { + "id": "2606.09961", + "title": "3SPO: State-Score-Supervised Policy Optimization for LLM Agents", + "url": "https://arxiv.org/abs/2606.09961", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Yu Han", + "Kailing Li", + "Yang Jiao", + "Yulin Dai", + "Yuqian Fu", + "Linhai Zhuo", + "Tianwen Qian" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "computer-use", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.09961", + "source": "arxiv", + "source_id": "arxiv:2606.09961", + "pdf_url": "https://arxiv.org/pdf/2606.09961", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.08172", + "title": "The Governance of Human-LLM Interaction: Safety Gating, Civility Steering, and Affective Default Lock-In", + "url": "https://arxiv.org/abs/2606.08172", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Manuele Reani", + "Hongjian Zhang", + "Hongyu Tian" + ], + "categories": [ + "cs.HC", + "cs.AI", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.08172", + "source": "arxiv", + "source_id": "arxiv:2606.08172", + "pdf_url": "https://arxiv.org/pdf/2606.08172", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.08162", + "title": "Silent Failure in LLM Agent Systems: The Entropy Principle and the Inevitable Disorder of Autonomous Agents", + "url": "https://arxiv.org/abs/2606.08162", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Dexing Liu" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "memory", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.08162", + "source": "arxiv", + "source_id": "arxiv:2606.08162", + "pdf_url": "https://arxiv.org/pdf/2606.08162", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.07836", + "title": "Agentic multi-fidelity learning of quasiparticle and excitonic properties", + "url": "https://arxiv.org/abs/2606.07836", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Arnab Neogi", + "Aaron Forde", + "Christopher A. Lane", + "Sergei Tretiak", + "Jian-Xin Zhu" + ], + "categories": [ + "cond-mat.mtrl-sci", + "cond-mat.stat-mech", + "cs.AI", + "physics.comp-ph", + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "workflow-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.07836", + "source": "arxiv", + "source_id": "arxiv:2606.07836", + "pdf_url": "https://arxiv.org/pdf/2606.07836", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18272", + "title": "Mitigating Anchoring Bias in LLM-Based Agents for Energy-Efficient 6G Autonomous Networks", + "url": "https://arxiv.org/abs/2606.18272", + "published": "2026-06-05", + "updated": "2026-06-18", + "authors": [ + "Hatim Chergui", + "Claudia Carballo González", + "Farhad Rezazadeh", + "Merouane Debbah" + ], + "categories": [ + "cs.NI", + "cs.AI", + "eess.SY" + ], + "topics": [ + "agent-safety", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.18272", + "source": "arxiv", + "source_id": "arxiv:2606.18272", + "pdf_url": "https://arxiv.org/pdf/2606.18272", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.05658", + "title": "Agent-Orchestrated Adaptive RAG: A Comparative Study on Structured and Multi-Hop Retrieval", + "url": "https://arxiv.org/abs/2606.05658", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Anuj Maharjan", + "Devinder Kaur", + "Richard Molyet" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.05658", + "source": "arxiv", + "source_id": "arxiv:2606.05658", + "pdf_url": "https://arxiv.org/pdf/2606.05658", + "primary_query": "rag-agent" + }, + { + "id": "2606.05622", + "title": "AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints", + "url": "https://arxiv.org/abs/2606.05622", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Jiayu Liu", + "Cheng Qian", + "Zhenhailong Wang", + "Bingxuan Li", + "Jiateng Liu", + "Heng Wang", + "Jeonghwan Kim", + "Yumeng Wang", + "Xiusi Chen", + "Yi R. Fung", + "Heng Ji" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.05622", + "source": "arxiv", + "source_id": "arxiv:2606.05622", + "pdf_url": "https://arxiv.org/pdf/2606.05622", + "primary_query": "planning-agent" + }, + { + "id": "2606.05436", + "title": "Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison", + "url": "https://arxiv.org/abs/2606.05436", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Alejandro Lozano", + "Keiko Ihara", + "Ping-Hao Yang", + "Carrie E. Robertson", + "Jennifer Stern", + "Allan Purdy", + "Hsiangkuo Yuan", + "Pengfei Zhang", + "Yulia Orlova", + "Olga Fermo", + "Jennifer Hranilovich", + "Fred Cohen", + "Todd J. Schwedt", + "Jenelle A. Jindal", + "Serena Yeung-Levy", + "Chia-Chun Chiang" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.05436", + "source": "arxiv", + "source_id": "arxiv:2606.05436", + "pdf_url": "https://arxiv.org/pdf/2606.05436", + "primary_query": "rag-agent" + }, + { + "id": "2606.03544", + "title": "SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems", + "url": "https://arxiv.org/abs/2606.03544", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Linyue Pan", + "Yaoming Zhu", + "Lin Qiu", + "Xuezhi Cao", + "Xunliang Cai" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.03544", + "source": "arxiv", + "source_id": "arxiv:2606.03544", + "pdf_url": "https://arxiv.org/pdf/2606.03544", + "primary_query": "language-agent" + }, + { + "id": "2606.03157", + "title": "ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models", + "url": "https://arxiv.org/abs/2606.03157", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Ruihui Hou", + "Siyi Zhu", + "Ziyue Huai", + "Guangya Yu", + "Yongqi Fan", + "Chunming Wang", + "Tong Ruan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.03157", + "source": "arxiv", + "source_id": "arxiv:2606.03157", + "pdf_url": "https://arxiv.org/pdf/2606.03157", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.01961", + "title": "AutoMedBench: Towards Medical AutoResearch with Agentic AI Models", + "url": "https://arxiv.org/abs/2606.01961", + "published": "2026-06-01", + "updated": "2026-06-03", + "authors": [ + "Junqi Liu", + "Selena Song", + "Yuhan Wang", + "Jiawei Mao", + "Hardy Chen", + "Xiaoke Huang", + "Tianhao Qi", + "Pengfei Guo", + "Yucheng Tang", + "Yufan He", + "Can Zhao", + "Andriy Myronenko", + "Dong Yang", + "Daguang Xu", + "Yuyin Zhou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.01961", + "source": "arxiv", + "source_id": "arxiv:2606.01961", + "pdf_url": "https://arxiv.org/pdf/2606.01961", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.01185", + "title": "\"Skill issues'': data-centric optimization of lakehouse agents", + "url": "https://arxiv.org/abs/2606.01185", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Nicole Rose Schneider", + "Davide Ghilardi", + "Giacomo Piccinini", + "Jacopo Tagliabue" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.01185", + "source": "arxiv", + "source_id": "arxiv:2606.01185", + "pdf_url": "https://arxiv.org/pdf/2606.01185", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.01138", + "title": "memorywire: A Vendor-Neutral Wire Format for Agent Memory Operations", + "url": "https://arxiv.org/abs/2606.01138", + "published": "2026-05-31", + "updated": "2026-06-03", + "authors": [ + "Thamilvendhan Munirathinam" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.01138", + "source": "arxiv", + "source_id": "arxiv:2606.01138", + "pdf_url": "https://arxiv.org/pdf/2606.01138", + "primary_query": "agent-memory" + }, + { + "id": "2606.01166", + "title": "BraveGuard: From Open-World Threats to Safer Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.01166", + "published": "2026-05-31", + "updated": "2026-06-02", + "authors": [ + "Yunhao Feng", + "Xiaohu Du", + "Xinhao Deng", + "Yifan Ding", + "Ming Wen", + "Yixu Wang", + "Yuxiang Xie", + "Baihui Zheng", + "Yingshui Tan", + "Yige Li", + "Yutao Wu", + "Kerui Cao", + "Wenke Huang", + "Yanming Guo", + "Xingjun Ma", + "Yu-Gang Jiang" + ], + "categories": [ + "cs.CR", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.01166", + "source": "arxiv", + "source_id": "arxiv:2606.01166", + "pdf_url": "https://arxiv.org/pdf/2606.01166", + "primary_query": "agent-safety" + }, + { + "id": "2606.00644", + "title": "ForeSci: Evaluating LLM Agents for Forward-Looking AI Research Judgment", + "url": "https://arxiv.org/abs/2606.00644", + "published": "2026-05-30", + "updated": "2026-06-04", + "authors": [ + "Qiuyu Tian", + "Haojie Yin", + "Yingce Xia", + "Youyong Kong", + "Zequn Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.00644", + "source": "arxiv", + "source_id": "arxiv:2606.00644", + "pdf_url": "https://arxiv.org/pdf/2606.00644", + "primary_query": "rag-agent" + }, + { + "id": "2605.31308", + "title": "TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories", + "url": "https://arxiv.org/abs/2605.31308", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Junjie Nian", + "Kang Chen", + "Ge Zhang", + "Yixin Cao", + "Yugang Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "embodied-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.31308", + "source": "arxiv", + "source_id": "arxiv:2605.31308", + "pdf_url": "https://arxiv.org/pdf/2605.31308", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.31075", + "title": "Task-Focused Memorization for Multimodal Agents", + "url": "https://arxiv.org/abs/2605.31075", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Tao Zou", + "Yichen He", + "Tian Qiu", + "Yuan Lin", + "Hang Li" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.31075", + "source": "arxiv", + "source_id": "arxiv:2605.31075", + "pdf_url": "https://arxiv.org/pdf/2605.31075", + "primary_query": "agent-memory" + }, + { + "id": "2605.31268", + "title": "Mellum2 Technical Report", + "url": "https://arxiv.org/abs/2605.31268", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Marko Kojic", + "Ivan Bondyrev", + "Aral de Moor", + "Joseph Shtok", + "Petr Borovlev", + "Kseniia Lysaniuk", + "Madeeswaran Kannan", + "Ivan Dolgov", + "Nikita Pavlichenko" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.31268", + "source": "arxiv", + "source_id": "arxiv:2605.31268", + "pdf_url": "https://arxiv.org/pdf/2605.31268", + "primary_query": "function-calling" + }, + { + "id": "2605.29653", + "title": "PTCG-Bench: Can LLM Agents Master Pokémon Trading Card Game?", + "url": "https://arxiv.org/abs/2605.29653", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Dongdong Hua", + "Yifei Sun", + "Renhong Huang", + "Feng Gao", + "Chunping Wang", + "Yang Yang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm" + ], + "arxiv_id": "2605.29653", + "source": "arxiv", + "source_id": "arxiv:2605.29653", + "pdf_url": "https://arxiv.org/pdf/2605.29653", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.29630", + "title": "Entity-Collision: A Stratified Protocol for Attributing Retrieval Lift in Agent Memory", + "url": "https://arxiv.org/abs/2605.29630", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Youwang Deng" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.29630", + "source": "arxiv", + "source_id": "arxiv:2605.29630", + "pdf_url": "https://arxiv.org/pdf/2605.29630", + "primary_query": "agent-memory" + }, + { + "id": "2606.07591", + "title": "ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research", + "url": "https://arxiv.org/abs/2606.07591", + "published": "2026-05-28", + "updated": "2026-07-03", + "authors": [ + "Wanghan Xu", + "Shuo Li", + "Tianlin Ye", + "Qinglong Cao", + "Yixin Chen", + "Hengjian Gao", + "Yiheng Wang", + "Qi Li", + "Kun Li", + "Sheng Xu", + "Shengdu Chai", + "Fangchen Yu", + "Xiangyu Zhao", + "Zhangrui Zhao", + "Weijie Ma", + "Zijie Guo", + "Koutian Wu", + "Haoyu Zhou", + "Haoxiang Yin", + "Lixue Cheng", + "Chaofan Hu", + "Haoxuan Li", + "Lu Mi", + "Xuxuan Xie", + "Yifan Zhou", + "Ruizhe Chen", + "Zhiwang Zhou", + "Xingjian Guo", + "Yuhao Zhou", + "Xuming He", + "Shengyuan Xu", + "Xinyu Gu", + "Jiamin Wu", + "Mianxin Liu", + "Chunfeng Song", + "Fenghua Ling", + "Dongzhan Zhou", + "Shixiang Tang", + "Yuqiang Li", + "Mao Su", + "Peng Ye", + "Siqi Sun", + "Bin Wang", + "Xue Yang", + "Zhenfei Yin", + "Tianfan Fu", + "Guangtao Zhai", + "Wanli Ouyang", + "Bo Zhang", + "Lei Bai", + "Wenlong Zhang" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.07591", + "source": "arxiv", + "source_id": "arxiv:2606.07591", + "pdf_url": "https://arxiv.org/pdf/2606.07591", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.28617", + "title": "LACUNA: Safe Agents as Recursive Program Holes", + "url": "https://arxiv.org/abs/2605.28617", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Yaoyu Zhao", + "Yichen Xu", + "Oliver Bračevac", + "Cao Nguyen Pham", + "Frank Zhengqing Wu", + "Martin Odersky" + ], + "categories": [ + "cs.AI", + "cs.PL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.28617", + "source": "arxiv", + "source_id": "arxiv:2605.28617", + "pdf_url": "https://arxiv.org/pdf/2605.28617", + "primary_query": "planning-agent" + }, + { + "id": "2605.28424", + "title": "Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2605.28424", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Jiapeng Zhu", + "Jianxiang Yu", + "Yibo Zhao", + "Chengcheng Han", + "Qi Gu", + "Xunliang Cai", + "Xiang Li", + "Weining Qian" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-safety", + "memory" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.28424", + "source": "arxiv", + "source_id": "arxiv:2605.28424", + "pdf_url": "https://arxiv.org/pdf/2605.28424", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.26497", + "title": "Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents", + "url": "https://arxiv.org/abs/2605.26497", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Peiran Wang", + "Ying Li", + "Yuan Tian" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.26497", + "source": "arxiv", + "source_id": "arxiv:2605.26497", + "pdf_url": "https://arxiv.org/pdf/2605.26497", + "primary_query": "agent-safety" + }, + { + "id": "2605.27123", + "title": "Rethinking Agentic RAG: Toward LLM-Driven Logical Retrieval Beyond Embeddings", + "url": "https://arxiv.org/abs/2605.27123", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Yuqi Zeng", + "Qixiang Deng", + "Yulei Wan", + "Ruiquan Jiang", + "Xiaoqing Zheng", + "Xuanjing Huang" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.27123", + "source": "arxiv", + "source_id": "arxiv:2605.27123", + "pdf_url": "https://arxiv.org/pdf/2605.27123", + "primary_query": "rag-agent" + }, + { + "id": "2605.26165", + "title": "Tool-Schema Compression Enables Agentic RAG Under Constrained Context Budgets", + "url": "https://arxiv.org/abs/2605.26165", + "published": "2026-05-24", + "updated": "2026-05-24", + "authors": [ + "Furkan Sakizli" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.26165", + "source": "arxiv", + "source_id": "arxiv:2605.26165", + "pdf_url": "https://arxiv.org/pdf/2605.26165", + "primary_query": "rag-agent" + }, + { + "id": "2605.23899", + "title": "From Raw Experience to Skill Consumption: A Systematic Study of Model-Generated Agent Skills", + "url": "https://arxiv.org/abs/2605.23899", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Zisu Huang", + "Jingwen Xu", + "Yifan Yang", + "Ziyang Gong", + "Qihao Yang", + "Muzhao Tian", + "Xiaohua Wang", + "Changze Lv", + "Xuemei Gao", + "Qi Dai", + "Bei Liu", + "Kai Qiu", + "Xue Yang", + "Dongdong Chen", + "Xiaoqing Zheng", + "Chong Luo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.23899", + "source": "arxiv", + "source_id": "arxiv:2605.23899", + "pdf_url": "https://arxiv.org/pdf/2605.23899", + "primary_query": "language-agent" + }, + { + "id": "2605.17075", + "title": "A Red Teaming Framework for Evaluating Robustness of AI-enabled Security Orchestration, Automation, and Response Systems", + "url": "https://arxiv.org/abs/2605.17075", + "published": "2026-05-16", + "updated": "2026-05-16", + "authors": [ + "Ayan Javeed Shaikh", + "Nathaniel D. Bastian", + "Ankit Shah" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.17075", + "source": "arxiv", + "source_id": "arxiv:2605.17075", + "pdf_url": "https://arxiv.org/pdf/2605.17075", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.14322", + "title": "Are Agents Ready to Teach? A Multi-Stage Benchmark for Real-World Teaching Workflows", + "url": "https://arxiv.org/abs/2605.14322", + "published": "2026-05-14", + "updated": "2026-05-20", + "authors": [ + "Zixin Chen", + "Peng Liu", + "Rui Sheng", + "Haobo Li", + "Jianhong Tu", + "Xiaodong Deng", + "Kashun Shum", + "Dayiheng Liu", + "Huamin Qu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.14322", + "source": "arxiv", + "source_id": "arxiv:2605.14322", + "pdf_url": "https://arxiv.org/pdf/2605.14322", + "primary_query": "language-agent" + }, + { + "id": "2605.11928", + "title": "When Simulation Lies: A Sim-to-Real Benchmark and Domain-Randomized RL Recipe for Tool-Use Agents", + "url": "https://arxiv.org/abs/2605.11928", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Xiaolin Zhou", + "Aojie Yuan", + "Zheng Luo", + "Zipeng Ling", + "Xixiao Pan", + "Yicheng Gao", + "Haiyue Zhang", + "Jiate Li", + "Shuli Jiang", + "Prince Zizhuang Wang", + "Zixuan Zhu", + "Jinbo Liu", + "Ryan A. Rossi", + "Hua Wei", + "Xiyang Hu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling", + "language-agent" + ], + "arxiv_id": "2605.11928", + "source": "arxiv", + "source_id": "arxiv:2605.11928", + "pdf_url": "https://arxiv.org/pdf/2605.11928", + "primary_query": "function-calling" + }, + { + "id": "2605.10870", + "title": "Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory", + "url": "https://arxiv.org/abs/2605.10870", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Mingxi Zou", + "Zhihan Guo", + "Langzhang Liang", + "Zhuo Wang", + "Qifan Wang", + "Qingsong Wen", + "Irwin King", + "Lizhen Qu", + "Zenglin Xu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.10870", + "source": "arxiv", + "source_id": "arxiv:2605.10870", + "pdf_url": "https://arxiv.org/pdf/2605.10870", + "primary_query": "language-agent" + }, + { + "id": "2605.08876", + "title": "OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents", + "url": "https://arxiv.org/abs/2605.08876", + "published": "2026-05-09", + "updated": "2026-06-07", + "authors": [ + "Xinyu Li", + "Ronghui Mu", + "Lin Li", + "Tianjin Huang", + "Gaojie Jin" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "computer-use", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.08876", + "source": "arxiv", + "source_id": "arxiv:2605.08876", + "pdf_url": "https://arxiv.org/pdf/2605.08876", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.03328", + "title": "LLM-ADAM: A Generalizable LLM Agent Framework for Pre-Print Anomaly Detection in Additive Manufacturing", + "url": "https://arxiv.org/abs/2605.03328", + "published": "2026-05-05", + "updated": "2026-05-05", + "authors": [ + "Ahmadreza Eslaminia", + "Chuhan Cai", + "Cameron Smith", + "Ruo-Syuan Mei", + "Shichen Li", + "Rajiv Malhotra", + "Klara Nahrstedt", + "Chenhui Shao" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.03328", + "source": "arxiv", + "source_id": "arxiv:2605.03328", + "pdf_url": "https://arxiv.org/pdf/2605.03328", + "primary_query": "planning-agent" + }, + { + "id": "2604.27092", + "title": "End-to-end autonomous scientific discovery on a real optical platform", + "url": "https://arxiv.org/abs/2604.27092", + "published": "2026-04-29", + "updated": "2026-04-29", + "authors": [ + "Shuxing Yang", + "Fujia Chen", + "Rui Zhao", + "Junyao Wu", + "Yize Wang", + "Haiyao Luo", + "Ning Han", + "Qiaolu Chen", + "Yuze Hu", + "Wenhao Li", + "Mingzhu Li", + "Hongsheng Chen", + "Yihao Yang" + ], + "categories": [ + "cs.AI", + "physics.optics" + ], + "topics": [ + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27092", + "source": "arxiv", + "source_id": "arxiv:2604.27092", + "pdf_url": "https://arxiv.org/pdf/2604.27092", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.20994", + "title": "Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models", + "url": "https://arxiv.org/abs/2604.20994", + "published": "2026-04-22", + "updated": "2026-04-22", + "authors": [ + "Yannis Belkhiter", + "Giulio Zizzo", + "Sergio Maffeis", + "Seshu Tirupathi", + "John D. Kelleher" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2604.20994", + "source": "arxiv", + "source_id": "arxiv:2604.20994", + "pdf_url": "https://arxiv.org/pdf/2604.20994", + "primary_query": "function-calling" + }, + { + "id": "2603.28900", + "title": "Robust Multi-Agent Reinforcement Learning for Small UAS Separation Assurance under GPS Degradation and Spoofing", + "url": "https://arxiv.org/abs/2603.28900", + "published": "2026-03-30", + "updated": "2026-03-30", + "authors": [ + "Alex Zongo", + "Filippos Fotiadis", + "Ufuk Topcu", + "Peng Wei" + ], + "categories": [ + "cs.RO", + "cs.AI", + "cs.LG", + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.28900", + "source": "arxiv", + "source_id": "arxiv:2603.28900", + "pdf_url": "https://arxiv.org/pdf/2603.28900", + "primary_query": "agent-safety" + }, + { + "id": "2603.25353", + "title": "SafeGuard ASF: SR Agentic Humanoid Robot System for Autonomous Industrial Safety", + "url": "https://arxiv.org/abs/2603.25353", + "published": "2026-03-26", + "updated": "2026-03-26", + "authors": [ + "Thanh Nguyen Canh", + "Thang Tran Viet", + "Thanh Tuan Tran", + "Ben Wei Lim" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-safety", + "embodied-agent", + "reasoning", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.25353", + "source": "arxiv", + "source_id": "arxiv:2603.25353", + "pdf_url": "https://arxiv.org/pdf/2603.25353", + "primary_query": "agent-safety" + }, + { + "id": "2603.19684", + "title": "TSegAgent: Zero-Shot Tooth Segmentation via Geometry-Aware Vision-Language Agents", + "url": "https://arxiv.org/abs/2603.19684", + "published": "2026-03-20", + "updated": "2026-06-23", + "authors": [ + "Shaojie Zhuang", + "Lu Yin", + "Guangshun Wei", + "Yunpeng Li", + "Xilu Wang", + "Yuanfeng Zhou" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.19684", + "source": "arxiv", + "source_id": "arxiv:2603.19684", + "pdf_url": "https://arxiv.org/pdf/2603.19684", + "primary_query": "language-agent" + }, + { + "id": "2603.17392", + "title": "Agentic Cognitive Profiling: Realigning Automated Alzheimer's Disease Detection with Clinical Construct Validity", + "url": "https://arxiv.org/abs/2603.17392", + "published": "2026-03-18", + "updated": "2026-03-18", + "authors": [ + "Jiawen Kang", + "Kun Li", + "Dongrui Han", + "Jinchao Li", + "Junan Li", + "Lingwei Meng", + "Xixin Wu", + "Helen Meng" + ], + "categories": [ + "cs.MA", + "cs.IR", + "q-bio.NC" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.17392", + "source": "arxiv", + "source_id": "arxiv:2603.17392", + "pdf_url": "https://arxiv.org/pdf/2603.17392", + "primary_query": "function-calling" + }, + { + "id": "2603.15666", + "title": "Compiled Memory: Not More Information, but More Precise Instructions for Language Agents", + "url": "https://arxiv.org/abs/2603.15666", + "published": "2026-03-12", + "updated": "2026-03-12", + "authors": [ + "James Rhodes", + "George Kang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.15666", + "source": "arxiv", + "source_id": "arxiv:2603.15666", + "pdf_url": "https://arxiv.org/pdf/2603.15666", + "primary_query": "language-agent" + }, + { + "id": "2603.00801", + "title": "The Synthetic Web: Adversarially-Curated Mini-Internets for Diagnosing Epistemic Weaknesses of Language Agents", + "url": "https://arxiv.org/abs/2603.00801", + "published": "2026-02-28", + "updated": "2026-02-28", + "authors": [ + "Shrey Shah", + "Levent Ozgur" + ], + "categories": [ + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.00801", + "source": "arxiv", + "source_id": "arxiv:2603.00801", + "pdf_url": "https://arxiv.org/pdf/2603.00801", + "primary_query": "language-agent" + }, + { + "id": "2602.21127", + "title": "\"Are You Sure?\": An Empirical Study of Human Perception Vulnerability in LLM-Driven Agentic Systems", + "url": "https://arxiv.org/abs/2602.21127", + "published": "2026-02-24", + "updated": "2026-02-24", + "authors": [ + "Xinfeng Li", + "Shenyu Dai", + "Kelong Zheng", + "Yue Xiao", + "Gelei Deng", + "Wei Dong", + "Xiaofeng Wang" + ], + "categories": [ + "cs.HC", + "cs.AI", + "cs.CR", + "cs.SI" + ], + "topics": [ + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.21127", + "source": "arxiv", + "source_id": "arxiv:2602.21127", + "pdf_url": "https://arxiv.org/pdf/2602.21127", + "primary_query": "agent-safety" + }, + { + "id": "2603.00131", + "title": "Thought Virus: Viral Misalignment via Subliminal Prompting in Multi-Agent Systems", + "url": "https://arxiv.org/abs/2603.00131", + "published": "2026-02-23", + "updated": "2026-02-23", + "authors": [ + "Moritz Weckbecker", + "Jonas Müller", + "Ben Hagag", + "Michael Mulet" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.00131", + "source": "arxiv", + "source_id": "arxiv:2603.00131", + "pdf_url": "https://arxiv.org/pdf/2603.00131", + "primary_query": "agent-safety" + }, + { + "id": "2602.19008", + "title": "Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks", + "url": "https://arxiv.org/abs/2602.19008", + "published": "2026-02-22", + "updated": "2026-02-22", + "authors": [ + "Wilson Y. Lee" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.19008", + "source": "arxiv", + "source_id": "arxiv:2602.19008", + "pdf_url": "https://arxiv.org/pdf/2602.19008", + "primary_query": "language-agent" + }, + { + "id": "2602.14234", + "title": "REDSearcher: A Scalable and Cost-Efficient Framework for Long-Horizon Search Agents", + "url": "https://arxiv.org/abs/2602.14234", + "published": "2026-02-15", + "updated": "2026-02-15", + "authors": [ + "Zheng Chu", + "Xiao Wang", + "Jack Hong", + "Huiming Fan", + "Yuqi Huang", + "Yue Yang", + "Guohai Xu", + "Chenxiao Zhao", + "Cheng Xiang", + "Shengchao Hu", + "Dongdong Kuang", + "Ming Liu", + "Bing Qin", + "Xing Yu" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2602.14234", + "source": "arxiv", + "source_id": "arxiv:2602.14234", + "pdf_url": "https://arxiv.org/pdf/2602.14234", + "primary_query": "function-calling" + }, + { + "id": "2602.14281", + "title": "MCPShield: A Security Cognition Layer for Adaptive Trust Calibration in Model Context Protocol Agents", + "url": "https://arxiv.org/abs/2602.14281", + "published": "2026-02-15", + "updated": "2026-02-24", + "authors": [ + "Zhenhong Zhou", + "Yuanhe Zhang", + "Hongwei Cai", + "Moayad Aloqaily", + "Ouns Bouachir", + "Linsey Pang", + "Prakhar Mehrotra", + "Kun Wang", + "Qingsong Wen" + ], + "categories": [ + "cs.CR", + "cs.CL" + ], + "topics": [ + "agent-safety", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.14281", + "source": "arxiv", + "source_id": "arxiv:2602.14281", + "pdf_url": "https://arxiv.org/pdf/2602.14281", + "primary_query": "agent-safety" + }, + { + "id": "2602.13665", + "title": "HyFunc: Accelerating LLM-based Function Calls for Agentic AI through Hybrid-Model Cascade and Dynamic Templating", + "url": "https://arxiv.org/abs/2602.13665", + "published": "2026-02-14", + "updated": "2026-02-14", + "authors": [ + "Weibin Liao", + "Jian-guang Lou", + "Haoyi Xiong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2602.13665", + "source": "arxiv", + "source_id": "arxiv:2602.13665", + "pdf_url": "https://arxiv.org/pdf/2602.13665", + "primary_query": "function-calling" + }, + { + "id": "2602.08082", + "title": "Spectral Guardrails for Agents in the Wild: Detecting Tool Use Hallucinations via Attention Topology", + "url": "https://arxiv.org/abs/2602.08082", + "published": "2026-02-08", + "updated": "2026-02-08", + "authors": [ + "Valentin Noël" + ], + "categories": [ + "cs.LG", + "cs.AI", + "eess.SP" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.08082", + "source": "arxiv", + "source_id": "arxiv:2602.08082", + "pdf_url": "https://arxiv.org/pdf/2602.08082", + "primary_query": "agent-safety" + }, + { + "id": "2602.10133", + "title": "AgentTrace: A Structured Logging Framework for Agent System Observability", + "url": "https://arxiv.org/abs/2602.10133", + "published": "2026-02-07", + "updated": "2026-02-07", + "authors": [ + "Adam AlSayyad", + "Kelvin Yuxiang Huang", + "Richik Pal" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.10133", + "source": "arxiv", + "source_id": "arxiv:2602.10133", + "pdf_url": "https://arxiv.org/pdf/2602.10133", + "primary_query": "agent-safety" + }, + { + "id": "2602.05386", + "title": "Spider-Sense: Intrinsic Risk Sensing for Efficient Agent Defense with Hierarchical Adaptive Screening", + "url": "https://arxiv.org/abs/2602.05386", + "published": "2026-02-05", + "updated": "2026-02-06", + "authors": [ + "Zhenxiong Yu", + "Zhi Yang", + "Zhiheng Jin", + "Shuhe Wang", + "Heng Zhang", + "Yanlin Fei", + "Lingfeng Zeng", + "Fangqi Lou", + "Shuo Zhang", + "Tu Hu", + "Jingping Liu", + "Rongze Chen", + "Xingyu Zhu", + "Kunyi Wang", + "Chaofa Yuan", + "Xin Guo", + "Zhaowei Liu", + "Feipeng Zhang", + "Jie Huang", + "Huacan Wang", + "Ronghao Chen", + "Liwen Zhang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.05386", + "source": "arxiv", + "source_id": "arxiv:2602.05386", + "pdf_url": "https://arxiv.org/pdf/2602.05386", + "primary_query": "agent-safety" + }, + { + "id": "2602.03117", + "title": "AgentDyn: Are Your Agent Security Defenses Deployable in Real-World Dynamic Environments?", + "url": "https://arxiv.org/abs/2602.03117", + "published": "2026-02-03", + "updated": "2026-05-07", + "authors": [ + "Hao Li", + "Ruoyao Wen", + "Shanghao Shi", + "Ning Zhang", + "Yevgeniy Vorobeychik", + "Chaowei Xiao" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.03117", + "source": "arxiv", + "source_id": "arxiv:2602.03117", + "pdf_url": "https://arxiv.org/pdf/2602.03117", + "primary_query": "agent-safety" + }, + { + "id": "2601.12988", + "title": "PaperGuide: Making Small Language-Model Paper-Reading Agents More Efficient", + "url": "https://arxiv.org/abs/2601.12988", + "published": "2026-01-19", + "updated": "2026-01-19", + "authors": [ + "Zijian Wang", + "Tiancheng Huang", + "Hanqi Li", + "Da Ma", + "Lu Chen", + "Kai Yu" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.12988", + "source": "arxiv", + "source_id": "arxiv:2601.12988", + "pdf_url": "https://arxiv.org/pdf/2601.12988", + "primary_query": "function-calling" + }, + { + "id": "2601.06606", + "title": "CEDAR: Context Engineering for Agentic Data Science", + "url": "https://arxiv.org/abs/2601.06606", + "published": "2026-01-10", + "updated": "2026-04-22", + "authors": [ + "Rishiraj Saha Roy", + "Chris Hinze", + "Luzian Hahn", + "Fabian Kuech" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "planning", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.06606", + "source": "arxiv", + "source_id": "arxiv:2601.06606", + "pdf_url": "https://arxiv.org/pdf/2601.06606", + "primary_query": "function-calling" + }, + { + "id": "2512.23747", + "title": "State-of-the-art Small Language Coder Model: Mify-Coder", + "url": "https://arxiv.org/abs/2512.23747", + "published": "2025-12-26", + "updated": "2025-12-26", + "authors": [ + "Abhinav Parmar", + "Abhisek Panigrahi", + "Abhishek Kumar Dwivedi", + "Abhishek Bhattacharya", + "Adarsh Ramachandra", + "Aditya Choudhary", + "Aditya Garg", + "Aditya Raj", + "Alankrit Bhatt", + "Alpesh Yadav", + "Anant Vishnu", + "Ananthu Pillai", + "Ankush Kumar", + "Aryan Patnaik", + "Aswatha Narayanan S", + "Avanish Raj Singh", + "Bhavya Shree Gadda", + "Brijesh Pankajbhai Kachhadiya", + "Buggala Jahnavi", + "Chidurala Nithin Krishna", + "Chintan Shah", + "Chunduru Akshaya", + "Debarshi Banerjee", + "Debrup Dey", + "Deepa R.", + "Deepika B G", + "Faiz ur Rahman", + "Gagan Gayari", + "Gudhi Jagadeesh Kumar Naidu", + "Gursimar Singh", + "Harshal Tyagi", + "Harshini K", + "James Mani Vathalloor", + "Jayarama Nettar", + "Jayashree Gajjam", + "Joe Walter Sugil George", + "Kamalakara Sri Krishna Tadepalli", + "Kamalkumar Rathinasamy", + "Karan Chaurasia", + "Karthikeyan S", + "Kashish Arora", + "Kaushal Desai", + "Khushboo Buwade", + "Kiran Manjrekar", + "Malikireddy Venkata Sai Likhitha", + "Manjunath A", + "Mitali Mahavir Bedmutha", + "Mohammed Rafee Tarafdar", + "Nikhil Tiwari", + "Nikitha K Gigi", + "Pavan Ravikumar", + "Pendyala Swarnanjali", + "Piyush Anand", + "Prakash Chandrasekar", + "Prasanna Bhalchandra Gawade", + "Prasanth Sivan", + "Preeti Khurana", + "Priyanshi Babbar", + "Rajab Ali Mondal", + "Rajesh Kumar Vissapragada", + "Rajeshwari Ganesan", + "Rajeswari Koppisetti", + "Ramjee R.", + "Ramkumar Thiruppathisamy", + "Rani G. S.", + "S Reka", + "Samarth Gupta", + "Sandeep Reddy Kothakota", + "Sarathy K", + "Sathyanarayana Sampath Kumar", + "Saurabh Kumar", + "Shashank Khasare", + "Shenbaga Devi Venkatesh Kumar", + "Shiva Rama Krishna Parvatham", + "Shoeb Shaikh", + "Shrishanmathi A", + "Shubham Pathak", + "Sree Samhita Koppaka", + "Sreenivasa Raghavan K S", + "Sreeram Venkatasubramanian", + "Suprabha Desai Bojja", + "Swetha R", + "Syed Ahmed", + "Chinmai Harshitha Thota", + "Tushar Yadav", + "Veeravelly Kusumitha", + "V V S S Prasanth Patnaik", + "Vidya Sri Sesetti", + "Vijayakeerthi K", + "Vikram Raj Bakshi", + "Vinay K K", + "Vinoth Kumar Loganathan", + "Vipin Tiwari", + "Vivek Kumar Shrivastav", + "V Venkata Sri Datta Charan", + "Wasim Akhtar Khan" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.23747", + "source": "arxiv", + "source_id": "arxiv:2512.23747", + "pdf_url": "https://arxiv.org/pdf/2512.23747", + "primary_query": "function-calling" + }, + { + "id": "2510.24645", + "title": "FunReason-MT Technical Report: Advanced Data Synthesis Solution for Real-world Multi-Turn Tool-use", + "url": "https://arxiv.org/abs/2510.24645", + "published": "2025-10-28", + "updated": "2025-11-16", + "authors": [ + "Zengzhuang Xu", + "Bingguang Hao", + "Zechuan Wang", + "Yuntao Wen", + "Xinyi Xu", + "Yang Liu", + "Long Chen", + "Dong Wang", + "Maolin Wang", + "Tong Zhao", + "Yicheng Chen", + "Cunyin Peng", + "Jinjie Gu", + "Leilei Gan", + "Xiangyu Zhao", + "Chenyi Zhuang", + "Shi Gu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.24645", + "source": "arxiv", + "source_id": "arxiv:2510.24645", + "pdf_url": "https://arxiv.org/pdf/2510.24645", + "primary_query": "function-calling" + }, + { + "id": "2510.04206", + "title": "AgentRL: Scaling Agentic Reinforcement Learning with a Multi-Turn, Multi-Task Framework", + "url": "https://arxiv.org/abs/2510.04206", + "published": "2025-10-05", + "updated": "2025-10-05", + "authors": [ + "Hanchen Zhang", + "Xiao Liu", + "Bowen Lv", + "Xueqiao Sun", + "Bohao Jing", + "Iat Long Iong", + "Zhenyu Hou", + "Zehan Qi", + "Hanyu Lai", + "Yifan Xu", + "Rui Lu", + "Hongning Wang", + "Jie Tang", + "Yuxiao Dong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.04206", + "source": "arxiv", + "source_id": "arxiv:2510.04206", + "pdf_url": "https://arxiv.org/pdf/2510.04206", + "primary_query": "function-calling" + }, + { + "id": "2509.13311", + "title": "Towards General Agentic Intelligence via Environment Scaling", + "url": "https://arxiv.org/abs/2509.13311", + "published": "2025-09-16", + "updated": "2025-09-16", + "authors": [ + "Runnan Fang", + "Shihao Cai", + "Baixuan Li", + "Jialong Wu", + "Guangyu Li", + "Wenbiao Yin", + "Xinyu Wang", + "Xiaobin Wang", + "Liangcai Su", + "Zhen Zhang", + "Shibin Wu", + "Zhengwei Tao", + "Yong Jiang", + "Pengjun Xie", + "Fei Huang", + "Jingren Zhou" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.13311", + "source": "arxiv", + "source_id": "arxiv:2509.13311", + "pdf_url": "https://arxiv.org/pdf/2509.13311", + "primary_query": "function-calling" + }, + { + "id": "2509.02494", + "title": "GridMind: LLMs-Powered Agents for Power System Analysis and Operations", + "url": "https://arxiv.org/abs/2509.02494", + "published": "2025-09-02", + "updated": "2025-09-02", + "authors": [ + "Hongwei Jin", + "Kibaek Kim", + "Jonghwan Kwon" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.02494", + "source": "arxiv", + "source_id": "arxiv:2509.02494", + "pdf_url": "https://arxiv.org/pdf/2509.02494", + "primary_query": "function-calling" + }, + { + "id": "2508.17094", + "title": "PowerChain: A Verifiable Agentic AI System for Automating Distribution Grid Analyses", + "url": "https://arxiv.org/abs/2508.17094", + "published": "2025-08-23", + "updated": "2025-10-21", + "authors": [ + "Emmanuel O. Badmus", + "Peng Sang", + "Dimitrios Stamoulis", + "Amritanshu Pandey" + ], + "categories": [ + "cs.AI", + "eess.SY" + ], + "topics": [ + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.17094", + "source": "arxiv", + "source_id": "arxiv:2508.17094", + "pdf_url": "https://arxiv.org/pdf/2508.17094", + "primary_query": "function-calling" + }, + { + "id": "2508.12685", + "title": "ToolACE-MT: Non-Autoregressive Generation for Agentic Multi-Turn Interaction", + "url": "https://arxiv.org/abs/2508.12685", + "published": "2025-08-18", + "updated": "2026-02-13", + "authors": [ + "Xingshan Zeng", + "Weiwen Liu", + "Lingzhi Wang", + "Liangyou Li", + "Fei Mi", + "Yasheng Wang", + "Lifeng Shang", + "Xin Jiang", + "Qun Liu" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.12685", + "source": "arxiv", + "source_id": "arxiv:2508.12685", + "pdf_url": "https://arxiv.org/pdf/2508.12685", + "primary_query": "function-calling" + }, + { + "id": "2507.20666", + "title": "MIMII-Agent: Leveraging LLMs with Function Calling for Relative Evaluation of Anomalous Sound Detection", + "url": "https://arxiv.org/abs/2507.20666", + "published": "2025-07-28", + "updated": "2025-07-28", + "authors": [ + "Harsh Purohit", + "Tomoya Nishida", + "Kota Dohi", + "Takashi Endo", + "Yohei Kawaguchi" + ], + "categories": [ + "eess.AS", + "cs.AI", + "cs.LG", + "cs.SD" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2507.20666", + "source": "arxiv", + "source_id": "arxiv:2507.20666", + "pdf_url": "https://arxiv.org/pdf/2507.20666", + "primary_query": "function-calling" + }, + { + "id": "2507.20395", + "title": "MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models", + "url": "https://arxiv.org/abs/2507.20395", + "published": "2025-07-27", + "updated": "2025-07-27", + "authors": [ + "Hafsteinn Einarsson" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2507.20395", + "source": "arxiv", + "source_id": "arxiv:2507.20395", + "pdf_url": "https://arxiv.org/pdf/2507.20395", + "primary_query": "function-calling" + }, + { + "id": "2607.06503", + "title": "Doomed from the Start: Early Abort of LLM Agent Episodes via a Recall-Controlled Probe Cascade", + "url": "https://arxiv.org/abs/2607.06503", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Kai Ruan", + "Zihe Huang", + "Ziqi Zhou", + "Qianshan Wei", + "Xuan Wang", + "Hao Sun" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.06503", + "source": "arxiv", + "source_id": "arxiv:2607.06503", + "pdf_url": "https://arxiv.org/pdf/2607.06503", + "primary_query": "llm-agent" + }, + { + "id": "2607.04729", + "title": "RustMizan: A Compilable, Contamination-Aware Benchmarking Framework for Rust Vulnerabilities", + "url": "https://arxiv.org/abs/2607.04729", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Tarek Elsayed", + "Shiping Yang", + "Eunsong Koh", + "Sanika Goyal", + "Vincent Huang", + "Paul Ngo", + "Nathan Young", + "Mohammad Omidvar Tehrani", + "Alvyn Kang", + "Arnell Kang", + "Zeyu Chen", + "Angélica Moreira", + "Xuan Feng", + "Angel X. Chang", + "Nick Sumner", + "Steven Y. Ko" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04729", + "source": "arxiv", + "source_id": "arxiv:2607.04729", + "pdf_url": "https://arxiv.org/pdf/2607.04729", + "primary_query": "llm-agent" + }, + { + "id": "2607.05577", + "title": "Narrative World Model: Narratology-Grounded Writer Memory for Long-Form Fiction", + "url": "https://arxiv.org/abs/2607.05577", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Mohammad Saifullah", + "Thomas Kornmaier", + "Taaha Kazi", + "Vasu Sharma", + "Aditya Sanjiv Kanade", + "Aanand Kumar Yadav" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use", + "world-model" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2607.05577", + "source": "arxiv", + "source_id": "arxiv:2607.05577", + "pdf_url": "https://arxiv.org/pdf/2607.05577", + "primary_query": "agent-memory" + }, + { + "id": "2607.05477", + "title": "Decision Protocols in Multi-Agent Large Language Model Conversations", + "url": "https://arxiv.org/abs/2607.05477", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Lars Benedikt Kaesberg" + ], + "categories": [ + "cs.MA", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.05477", + "source": "arxiv", + "source_id": "arxiv:2607.05477", + "pdf_url": "https://arxiv.org/pdf/2607.05477", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.04235", + "title": "Spinning Straw into Gold: Relabeling LLM Agent Trajectories in Hindsight for Successful Demonstrations", + "url": "https://arxiv.org/abs/2607.04235", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Zichao Li", + "Gang Wu", + "Zichao Wang", + "Ruiyi Zhang", + "Wanrong Zhu", + "Ryan A. Rossi", + "Vlad I Morariu", + "Jihyung Kil" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "planning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04235", + "source": "arxiv", + "source_id": "arxiv:2607.04235", + "pdf_url": "https://arxiv.org/pdf/2607.04235", + "primary_query": "llm-agent" + }, + { + "id": "2607.01764", + "title": "Mastermind: Strategy-grounded Learning for Repository-Scale Vulnerability Reproduction", + "url": "https://arxiv.org/abs/2607.01764", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Mingzhe Du", + "Luu Anh Tuan", + "Tianyi Wu", + "Renyang Liu", + "Zhijiang Guo", + "Dong Huang", + "See-Kiong Ng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01764", + "source": "arxiv", + "source_id": "arxiv:2607.01764", + "pdf_url": "https://arxiv.org/pdf/2607.01764", + "primary_query": "llm-agent" + }, + { + "id": "2607.02116", + "title": "ContextNest: Verifiable Context Governance for Autonomous AI Agent", + "url": "https://arxiv.org/abs/2607.02116", + "published": "2026-07-02", + "updated": "2026-07-06", + "authors": [ + "Misha Sulpovar", + "Benn R. Konsynski", + "Qaish Kanchwala", + "Gabe Goodhart" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2607.02116", + "source": "arxiv", + "source_id": "arxiv:2607.02116", + "pdf_url": "https://arxiv.org/pdf/2607.02116", + "primary_query": "ai-agent" + }, + { + "id": "2607.02389", + "title": "Steerability via constraints: a substrate for scalable oversight of coding agents", + "url": "https://arxiv.org/abs/2607.02389", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Thomas Winninger" + ], + "categories": [ + "cs.AI", + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02389", + "source": "arxiv", + "source_id": "arxiv:2607.02389", + "pdf_url": "https://arxiv.org/pdf/2607.02389", + "primary_query": "coding-agent" + }, + { + "id": "2607.02802", + "title": "Seduced by the Narrative: Assessing Rule Adherence in Semi-Open Textual Sandboxes", + "url": "https://arxiv.org/abs/2607.02802", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Weiying Chen", + "Junlong Shen", + "Zhanyuan Guo", + "Xiaoou Zhou" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.02802", + "source": "arxiv", + "source_id": "arxiv:2607.02802", + "pdf_url": "https://arxiv.org/pdf/2607.02802", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.01846", + "title": "CLAP: Closed-Loop Training, Evaluation, and Release Control for Domain Agent Post-training", + "url": "https://arxiv.org/abs/2607.01846", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Fangfei Li", + "Chenyang Zhao", + "Long Wang", + "Feng Tian", + "Zhiyue Zheng", + "Lv Guo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.01846", + "source": "arxiv", + "source_id": "arxiv:2607.01846", + "pdf_url": "https://arxiv.org/pdf/2607.01846", + "primary_query": "rag-agent" + }, + { + "id": "2607.01136", + "title": "Skills Are Not Islands: Measuring Dependency and Risk in Agent Skill Supply Chains", + "url": "https://arxiv.org/abs/2607.01136", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Changguo Jia", + "Tianqi Zhao", + "Runzhi He", + "Minghui Zhou" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01136", + "source": "arxiv", + "source_id": "arxiv:2607.01136", + "pdf_url": "https://arxiv.org/pdf/2607.01136", + "primary_query": "llm-agent" + }, + { + "id": "2607.00339", + "title": "TRACE: State-Aware Query Processing over Temporal Evidence Graphs for Conversational Data", + "url": "https://arxiv.org/abs/2607.00339", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Maolin Wang", + "Yu Wang", + "Zichun Liu", + "Baiyuan Qiu", + "Chenbin Zhang", + "Jiguang Shen", + "Haoran Yang", + "Hao Miao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.00339", + "source": "arxiv", + "source_id": "arxiv:2607.00339", + "pdf_url": "https://arxiv.org/pdf/2607.00339", + "primary_query": "ai-agent" + }, + { + "id": "2607.02605", + "title": "A Survey of LLM-Driven Penetration Testing: Taxonomy, Co-Evolution, and Open Challenges", + "url": "https://arxiv.org/abs/2607.02605", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Zheyuan He", + "Jiaxun Dong", + "Zihao Li", + "Ting Chen", + "Gelei Deng", + "Feng Luo", + "Jinkun Ji", + "Yuanlong Cao", + "Xiapu Luo" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2607.02605", + "source": "arxiv", + "source_id": "arxiv:2607.02605", + "pdf_url": "https://arxiv.org/pdf/2607.02605", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.31518", + "title": "Design and Implementation of Agentic Orchestrations and Orchestration of Agents", + "url": "https://arxiv.org/abs/2606.31518", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Stefanie Rinderle-Ma", + "Juergen Mangler", + "Johannes Loebbecke", + "Dominik Voigt", + "Nataliia Klievtsova", + "Matthias Ehrendorfer" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.31518", + "source": "arxiv", + "source_id": "arxiv:2606.31518", + "pdf_url": "https://arxiv.org/pdf/2606.31518", + "primary_query": "ai-agent" + }, + { + "id": "2606.31023", + "title": "Certified Speculative Execution for Untrusted AI Agents", + "url": "https://arxiv.org/abs/2606.31023", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Chenyu Zhou", + "Qiliang Jiang", + "Shuning Wu", + "Xu Zhou" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-safety", + "reasoning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.31023", + "source": "arxiv", + "source_id": "arxiv:2606.31023", + "pdf_url": "https://arxiv.org/pdf/2606.31023", + "primary_query": "ai-agent" + }, + { + "id": "2606.31613", + "title": "Robust Autonomous UAV Landing on Maritime Platforms via Multimodal Agentic AI and Active Wave Compensation", + "url": "https://arxiv.org/abs/2606.31613", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Francisco S. Neves", + "Pedro N. Pereira", + "Raul D. S. G. Campilho", + "Andry M. Pinto" + ], + "categories": [ + "cs.CV", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "world-model" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.31613", + "source": "arxiv", + "source_id": "arxiv:2606.31613", + "pdf_url": "https://arxiv.org/pdf/2606.31613", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31392", + "title": "ReGRPO: Reflection-Augmented Policy Optimization for Tool-Using Agents", + "url": "https://arxiv.org/abs/2606.31392", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Binjie Zhang", + "Mike Zheng Shou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.31392", + "source": "arxiv", + "source_id": "arxiv:2606.31392", + "pdf_url": "https://arxiv.org/pdf/2606.31392", + "primary_query": "tool-use" + }, + { + "id": "2606.31461", + "title": "CSTrader: A Testbed for Language-Grounded Trading in a Community-Driven Virtual Asset Market", + "url": "https://arxiv.org/abs/2606.31461", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yao Shi", + "Kingfung Luo", + "Nan Tang", + "Yuyu Luo" + ], + "categories": [ + "cs.AI", + "cs.CE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31461", + "source": "arxiv", + "source_id": "arxiv:2606.31461", + "pdf_url": "https://arxiv.org/pdf/2606.31461", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31039", + "title": "Truth or Sophistry? LoFa: A Benchmark for LLM Robustness Against Logical Fallacies", + "url": "https://arxiv.org/abs/2606.31039", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Xudong Shen", + "Li Yuan", + "Ye Chen", + "Xin Wu", + "Yi Cai", + "Zhiyong Wu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31039", + "source": "arxiv", + "source_id": "arxiv:2606.31039", + "pdf_url": "https://arxiv.org/pdf/2606.31039", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29722", + "title": "Attraction, Not Adaptation: How AI Agent Communities Develop Distinct Linguistic Identities", + "url": "https://arxiv.org/abs/2606.29722", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Daming Li", + "Simeng Han", + "Can Meng", + "Wanyu Lei", + "Jialu Zhang" + ], + "categories": [ + "cs.SI" + ], + "topics": [ + "computer-use", + "multi-agent", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.29722", + "source": "arxiv", + "source_id": "arxiv:2606.29722", + "pdf_url": "https://arxiv.org/pdf/2606.29722", + "primary_query": "ai-agent" + }, + { + "id": "2606.30531", + "title": "Entity Binding Failures in Tool-Augmented Agents", + "url": "https://arxiv.org/abs/2606.30531", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Rahul Suresh Babu", + "Shashank Indukuri" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.30531", + "source": "arxiv", + "source_id": "arxiv:2606.30531", + "pdf_url": "https://arxiv.org/pdf/2606.30531", + "primary_query": "tool-use" + }, + { + "id": "2606.29871", + "title": "AI Training Manager: Bounded Closed-Loop Control of Adaptive Training Recipes", + "url": "https://arxiv.org/abs/2606.29871", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Anjali Rao", + "Nikhil Kamalkumar Advani" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.29871", + "source": "arxiv", + "source_id": "arxiv:2606.29871", + "pdf_url": "https://arxiv.org/pdf/2606.29871", + "primary_query": "coding-agent" + }, + { + "id": "2606.30479", + "title": "COHORT: Collaborative Orchestration for Hardening via Offensive Replay on Emulated Topologies", + "url": "https://arxiv.org/abs/2606.30479", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Chen Frydman", + "Aviram Zilberman", + "Rubin Krief", + "Abed Showgan", + "Andres Murillo", + "Sekiya Motoyoshi", + "Asaf Shabtai", + "Yuval Elovici", + "Rami Puzis" + ], + "categories": [ + "cs.NI", + "cs.AI", + "cs.CR", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30479", + "source": "arxiv", + "source_id": "arxiv:2606.30479", + "pdf_url": "https://arxiv.org/pdf/2606.30479", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29033", + "title": "Human-in-the-Loop Nugget Annotation for Accountable LLM-as-a-Judge Evaluations", + "url": "https://arxiv.org/abs/2606.29033", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Laura Dietz" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.29033", + "source": "arxiv", + "source_id": "arxiv:2606.29033", + "pdf_url": "https://arxiv.org/pdf/2606.29033", + "primary_query": "ai-agent" + }, + { + "id": "2606.27936", + "title": "Agentic AI-Powered Re-Identification: An Emerging, Scalable Threat to Mobility Microdata Privacy", + "url": "https://arxiv.org/abs/2606.27936", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Oscar Thees", + "Roman Müller", + "Matthias Templ" + ], + "categories": [ + "cs.CR", + "cs.AI", + "stat.AP" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.27936", + "source": "arxiv", + "source_id": "arxiv:2606.27936", + "pdf_url": "https://arxiv.org/pdf/2606.27936", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26790", + "title": "OPID: On-Policy Skill Distillation for Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.26790", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Shuo Yang", + "Jinyang Wu", + "Zhengxi Lu", + "Yuhao Shen", + "Fan Zhang", + "Lang Feng", + "Shuai Zhang", + "Haoran Luo", + "Zheng Lian", + "Zhengqi Wen", + "Jianhua Tao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.26790", + "source": "arxiv", + "source_id": "arxiv:2606.26790", + "pdf_url": "https://arxiv.org/pdf/2606.26790", + "primary_query": "language-agent" + }, + { + "id": "2606.27443", + "title": "When Does Personality Composition Matter for Multi-Agent LLM Teams?", + "url": "https://arxiv.org/abs/2606.27443", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Aryan Keluskar", + "Amrita Bhattacharjee", + "Huan Liu" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "coding-agent", + "multi-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.27443", + "source": "arxiv", + "source_id": "arxiv:2606.27443", + "pdf_url": "https://arxiv.org/pdf/2606.27443", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.27409", + "title": "Delayed Verification Destabilizes Multi-Agent LLM Belief: Instability Thresholds and Optimal Corrector Placement", + "url": "https://arxiv.org/abs/2606.27409", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Igor Itkin" + ], + "categories": [ + "cs.MA", + "cs.CL", + "cs.LG", + "eess.SY" + ], + "topics": [ + "multi-agent", + "reasoning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.27409", + "source": "arxiv", + "source_id": "arxiv:2606.27409", + "pdf_url": "https://arxiv.org/pdf/2606.27409", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.26289", + "title": "Augmentation with Dilution: A Large-Scale Empirical Study of Human Contributor Ecosystems After AI Coding Agent Adoption", + "url": "https://arxiv.org/abs/2606.26289", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Weixing Zhang", + "Bowen Jiang", + "Anne Koziolek" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "coding-agent", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2606.26289", + "source": "arxiv", + "source_id": "arxiv:2606.26289", + "pdf_url": "https://arxiv.org/pdf/2606.26289", + "primary_query": "ai-agent" + }, + { + "id": "2606.25996", + "title": "Autodata: An agentic data scientist to create high quality synthetic data", + "url": "https://arxiv.org/abs/2606.25996", + "published": "2026-06-24", + "updated": "2026-07-04", + "authors": [ + "Ilia Kulikov", + "Chenxi Whitehouse", + "Tianhao Wu", + "Yixin Nie", + "Swarnadeep Saha", + "Eryk Helenowski", + "Weizhe Yuan", + "Olga Golovneva", + "Jack Lanchantin", + "Yoram Bachrach", + "Jakob Foerster", + "Xian Li", + "Han Fang", + "Sainbayar Sukhbaatar", + "Jason Weston" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.25996", + "source": "arxiv", + "source_id": "arxiv:2606.25996", + "pdf_url": "https://arxiv.org/pdf/2606.25996", + "primary_query": "ai-agent" + }, + { + "id": "2606.25836", + "title": "AI Snitches Get Glitches: Towards Evading Agentic Surveillance", + "url": "https://arxiv.org/abs/2606.25836", + "published": "2026-06-24", + "updated": "2026-06-26", + "authors": [ + "Hyejun Jeong", + "Dzung Pham", + "Amir Houmansadr", + "Eugene Bagdasarian" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.25836", + "source": "arxiv", + "source_id": "arxiv:2606.25836", + "pdf_url": "https://arxiv.org/pdf/2606.25836", + "primary_query": "ai-agent" + }, + { + "id": "2606.25332", + "title": "Decoupling Reconnaissance and Exploitation: Measuring the Capability Boundaries of LLM-Based Web Penetration Testing", + "url": "https://arxiv.org/abs/2606.25332", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Liwei Yu", + "Shuo Li", + "Ming Zhou", + "Ge Chu", + "Yan Guo" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25332", + "source": "arxiv", + "source_id": "arxiv:2606.25332", + "pdf_url": "https://arxiv.org/pdf/2606.25332", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24530", + "title": "NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?", + "url": "https://arxiv.org/abs/2606.24530", + "published": "2026-06-23", + "updated": "2026-07-06", + "authors": [ + "Yuru Wang", + "Lejun Cheng", + "Yuxin Zuo", + "Sihang Zeng", + "Bingxiang He", + "Che Jiang", + "Junlin Yang", + "Yuchong Wang", + "Kaikai Zhao", + "Weifeng Huang", + "Kai Tian", + "Zhenzhao Yuan", + "Jincheng Zhong", + "Weizhi Wang", + "Ning Ding", + "Bowen Zhou", + "Kaiyan Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.24530", + "source": "arxiv", + "source_id": "arxiv:2606.24530", + "pdf_url": "https://arxiv.org/pdf/2606.24530", + "primary_query": "coding-agent" + }, + { + "id": "2606.24370", + "title": "When Helpfulness Overrides Causal Caution: Context-Dependent Suppression and Recovery in LLMs", + "url": "https://arxiv.org/abs/2606.24370", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Hiroshi Okumura" + ], + "categories": [ + "cs.AI", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "reasoning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24370", + "source": "arxiv", + "source_id": "arxiv:2606.24370", + "pdf_url": "https://arxiv.org/pdf/2606.24370", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22916", + "title": "Intent-Governed Tool Authorization for AI Agents", + "url": "https://arxiv.org/abs/2606.22916", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Genliang Zhu", + "Chu Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "ai-agent", + "tool-use" + ], + "arxiv_id": "2606.22916", + "source": "arxiv", + "source_id": "arxiv:2606.22916", + "pdf_url": "https://arxiv.org/pdf/2606.22916", + "primary_query": "ai-agent" + }, + { + "id": "2606.23654", + "title": "EnterpriseClawBench: Benchmarking Agents from Real Workplace Sessions", + "url": "https://arxiv.org/abs/2606.23654", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Jincheng Zhong", + "Weizhi Wang", + "Che Jiang", + "Kai Tian", + "Zhenzhao Yuan", + "Junlin Yang", + "Dianqiao Lei", + "Kaiyan Zhang" + ], + "categories": [ + "cs.CL", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.23654", + "source": "arxiv", + "source_id": "arxiv:2606.23654", + "pdf_url": "https://arxiv.org/pdf/2606.23654", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.23277", + "title": "GIF: Locally Sound Geometric Information Flow Control for LLMs", + "url": "https://arxiv.org/abs/2606.23277", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Adam Storek", + "Nikolaus Holzer", + "Zhuo Zhang", + "Suman Jana" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.23277", + "source": "arxiv", + "source_id": "arxiv:2606.23277", + "pdf_url": "https://arxiv.org/pdf/2606.23277", + "primary_query": "tool-use" + }, + { + "id": "2606.23112", + "title": "Self-Evolution for Multi-Turn Tool-Calling Agents via Divergence-Point Preference Learning", + "url": "https://arxiv.org/abs/2606.23112", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Jiaqiang Tang" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.23112", + "source": "arxiv", + "source_id": "arxiv:2606.23112", + "pdf_url": "https://arxiv.org/pdf/2606.23112", + "primary_query": "tool-use" + }, + { + "id": "2606.23797", + "title": "From Task-Guided Conversational Graphs to Goal-Oriented Dialogue Runtimes", + "url": "https://arxiv.org/abs/2606.23797", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Mariano Garralda-Barrio" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.23797", + "source": "arxiv", + "source_id": "arxiv:2606.23797", + "pdf_url": "https://arxiv.org/pdf/2606.23797", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22329", + "title": "BabelJudge: Measuring LLM-as-a-Judge Reliability Across Languages and Agent Trajectories", + "url": "https://arxiv.org/abs/2606.22329", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Shreyas KC" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.22329", + "source": "arxiv", + "source_id": "arxiv:2606.22329", + "pdf_url": "https://arxiv.org/pdf/2606.22329", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.22504", + "title": "Lingering Authority: Revocable Resource-and-Effect Capabilities for Coding Agents", + "url": "https://arxiv.org/abs/2606.22504", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Igor Santos-Grueiro" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.22504", + "source": "arxiv", + "source_id": "arxiv:2606.22504", + "pdf_url": "https://arxiv.org/pdf/2606.22504", + "primary_query": "coding-agent" + }, + { + "id": "2606.22337", + "title": "Theorist Toolbox: Tools for Agent Based LLM-assisted economic theory Research", + "url": "https://arxiv.org/abs/2606.22337", + "published": "2026-06-21", + "updated": "2026-06-23", + "authors": [ + "Moran Koren" + ], + "categories": [ + "econ.TH", + "cs.GT", + "econ.GN" + ], + "topics": [ + "computer-use", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.22337", + "source": "arxiv", + "source_id": "arxiv:2606.22337", + "pdf_url": "https://arxiv.org/pdf/2606.22337", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22692", + "title": "VISTA Architect: A graph database-oriented health AI system demonstrated in multidisciplinary tumor boards", + "url": "https://arxiv.org/abs/2606.22692", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Tuomo Kiiskinen", + "Jason Fries", + "Philip Adamson", + "David Wu", + "Timothy John Ellis-Caleo", + "Aaron Fanous", + "Balasubramanian Narasimhan", + "Joel Neal", + "Sylvia Plevritis", + "Manuel A. Rivas" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.DB", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.22692", + "source": "arxiv", + "source_id": "arxiv:2606.22692", + "pdf_url": "https://arxiv.org/pdf/2606.22692", + "primary_query": "rag-agent" + }, + { + "id": "2606.21955", + "title": "From RAN Control to Agentic Intelligence: Architecture and Vision for Energy Efficient AI-RAN", + "url": "https://arxiv.org/abs/2606.21955", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Sabrine Aroua", + "Alexis I. Aravanis", + "Ilias Chatzistefanidis", + "Hamza Abbar", + "Anh-Khoa Dang", + "Anastasios Giovanidis", + "Salah-Eddine El Ayoubi", + "Stephane Senecal", + "Martha Vlachou Konchylaki", + "Navid Nikaein" + ], + "categories": [ + "cs.NI", + "cs.AI" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.21955", + "source": "arxiv", + "source_id": "arxiv:2606.21955", + "pdf_url": "https://arxiv.org/pdf/2606.21955", + "primary_query": "agentic-ai" + }, + { + "id": "2606.21894", + "title": "Skills for the future software profession: beyond agentic AI!", + "url": "https://arxiv.org/abs/2606.21894", + "published": "2026-06-20", + "updated": "2026-06-23", + "authors": [ + "Sungmin Kang", + "Baishakhi Ray", + "Abhik Roychoudhury" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2606.21894", + "source": "arxiv", + "source_id": "arxiv:2606.21894", + "pdf_url": "https://arxiv.org/pdf/2606.21894", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20978", + "title": "How Should Agents Read Demonstrations? Hierarchical Structure Beats Flat Action Logs", + "url": "https://arxiv.org/abs/2606.20978", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Honjar Xing", + "Jefferson Lin", + "Henry Lieberman" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.20978", + "source": "arxiv", + "source_id": "arxiv:2606.20978", + "pdf_url": "https://arxiv.org/pdf/2606.20978", + "primary_query": "planning-agent" + }, + { + "id": "2606.20729", + "title": "LLM-Guided Test-Time Discovery of Quantum-Chemical Approximation Algorithms", + "url": "https://arxiv.org/abs/2606.20729", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Masaya Hagai", + "Yuta Suzuki", + "Tomoya Murata", + "Shuhei Kurita", + "Masaki Adachi" + ], + "categories": [ + "physics.chem-ph", + "cond-mat.mtrl-sci", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.20729", + "source": "arxiv", + "source_id": "arxiv:2606.20729", + "pdf_url": "https://arxiv.org/pdf/2606.20729", + "primary_query": "agentic-ai" + }, + { + "id": "2606.19416", + "title": "MortarBench: Evaluating Mortgage Loan Origination Agents", + "url": "https://arxiv.org/abs/2606.19416", + "published": "2026-06-17", + "updated": "2026-06-22", + "authors": [ + "Matthew Toles", + "Yunan Lu", + "Manav Munjal", + "Bojun Liu", + "Yuanhao Deng", + "Stephanie Selig", + "Derek Rindner", + "Cheng Li", + "Zhou Yu" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.19416", + "source": "arxiv", + "source_id": "arxiv:2606.19416", + "pdf_url": "https://arxiv.org/pdf/2606.19416", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18550", + "title": "The Gate Is Only as Honest as Its Contracts: ContractGuard for the Contract Layer of Risk-Aware Causal Gating", + "url": "https://arxiv.org/abs/2606.18550", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Laxmipriya Ganesh Iyer", + "Rahul Suresh Babu" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.18550", + "source": "arxiv", + "source_id": "arxiv:2606.18550", + "pdf_url": "https://arxiv.org/pdf/2606.18550", + "primary_query": "tool-use" + }, + { + "id": "2606.19616", + "title": "Before the Pull Request: Mining Multi-Agent Coordination", + "url": "https://arxiv.org/abs/2606.19616", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Dipankar Sarkar" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.MA" + ], + "topics": [ + "coding-agent", + "multi-agent", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.19616", + "source": "arxiv", + "source_id": "arxiv:2606.19616", + "pdf_url": "https://arxiv.org/pdf/2606.19616", + "primary_query": "coding-agent" + }, + { + "id": "2606.18890", + "title": "Skill-Guided Continuation Distillation for GUI Agents", + "url": "https://arxiv.org/abs/2606.18890", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Zhimin Fan", + "Hongwei Yu", + "Yeqing Shen", + "Haolong Yan", + "Guozhen Peng", + "Tianhao Peng", + "Yudong Zhang", + "Xiaowen Zhang", + "Kaijun Tan", + "Zheng Ge", + "Xiangyu Zhang", + "Daxin Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.18890", + "source": "arxiv", + "source_id": "arxiv:2606.18890", + "pdf_url": "https://arxiv.org/pdf/2606.18890", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.19602", + "title": "Configurable Clinical Information Extraction with Agentic RAG: What Works, What Breaks, and Why", + "url": "https://arxiv.org/abs/2606.19602", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Osman Alperen Çinar-Koraş", + "Marie Bauer", + "Sameh Khattab", + "Merlin Engelke", + "Moon Kim", + "Stephan Settelmeier", + "Shigeyasu Sugawara", + "Fabian Freisleben", + "Felix Nensa", + "Jens Kleesiek" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.19602", + "source": "arxiv", + "source_id": "arxiv:2606.19602", + "pdf_url": "https://arxiv.org/pdf/2606.19602", + "primary_query": "rag-agent" + }, + { + "id": "2606.30658", + "title": "Agentic AI Enhances Physician Trust in Clinical Decision Making", + "url": "https://arxiv.org/abs/2606.30658", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Zhiling Yan", + "Zhe Fang", + "David J King", + "Ann Pongsakul", + "Eashan Adhikarla", + "Hui Ren", + "Sunyang Fu", + "Quanzheng Li", + "Lifang He", + "Xiang Li", + "Hongfang Liu", + "Yonghui Wu", + "Lichao Sun" + ], + "categories": [ + "cs.CY", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.30658", + "source": "arxiv", + "source_id": "arxiv:2606.30658", + "pdf_url": "https://arxiv.org/pdf/2606.30658", + "primary_query": "agentic-ai" + }, + { + "id": "2606.17628", + "title": "OPD-Evolver: Cultivating Holistic Agent Evolver via On-Policy Distillation", + "url": "https://arxiv.org/abs/2606.17628", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Guibin Zhang", + "Xun Xu", + "Yanwei Yue", + "Zikun Su", + "Wangchunshu Zhou", + "Xiaobin Hu", + "Shuicheng Yan" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.17628", + "source": "arxiv", + "source_id": "arxiv:2606.17628", + "pdf_url": "https://arxiv.org/pdf/2606.17628", + "primary_query": "agent-memory" + }, + { + "id": "2606.20724", + "title": "When Web Agents Finish but Still Fail: Reproducible Triggers and Trace Diagnostics for Parallel Web Exploration", + "url": "https://arxiv.org/abs/2606.20724", + "published": "2026-06-16", + "updated": "2026-06-29", + "authors": [ + "Aagam Sogani", + "Botao Rui", + "Swetha Vaidyanathan", + "Rishi Agarwal", + "Minghao Yan", + "Shivaram Venkataraman" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20724", + "source": "arxiv", + "source_id": "arxiv:2606.20724", + "pdf_url": "https://arxiv.org/pdf/2606.20724", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.18448", + "title": "VISUALSKILL: Multimodal Skills for Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.18448", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Ziyan Jiang", + "Li An", + "Yujian Liu", + "Jiabao Ji", + "Qiucheng Wu", + "Jacob Andreas", + "Yang Zhang", + "Shiyu Chang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.18448", + "source": "arxiv", + "source_id": "arxiv:2606.18448", + "pdf_url": "https://arxiv.org/pdf/2606.18448", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.15139", + "title": "Self-Driving Negotiator: An interactive, verifiable benchmark for social negotiation and theory of mind under hidden intent", + "url": "https://arxiv.org/abs/2606.15139", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Ashutosh Kumar" + ], + "categories": [ + "cs.GT", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.15139", + "source": "arxiv", + "source_id": "arxiv:2606.15139", + "pdf_url": "https://arxiv.org/pdf/2606.15139", + "primary_query": "language-agent" + }, + { + "id": "2606.15385", + "title": "Reward Hacking in Language Model Agents: Revisiting AI Safety Gridworlds", + "url": "https://arxiv.org/abs/2606.15385", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Ömer Veysel Çağatan", + "Xuandong Zhao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.15385", + "source": "arxiv", + "source_id": "arxiv:2606.15385", + "pdf_url": "https://arxiv.org/pdf/2606.15385", + "primary_query": "agent-safety" + }, + { + "id": "2606.14989", + "title": "Hierarchical Generative Agents for Simulating Sequential Human Behavior", + "url": "https://arxiv.org/abs/2606.14989", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Maria G. Mendoza", + "Lucas Waldburger", + "Jin Lee", + "Shankar Sastry" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "embodied-agent", + "planning", + "reasoning", + "world-model" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.14989", + "source": "arxiv", + "source_id": "arxiv:2606.14989", + "pdf_url": "https://arxiv.org/pdf/2606.14989", + "primary_query": "planning-agent" + }, + { + "id": "2606.13220", + "title": "LLM-as-an-Investigator: Evidence-First Reasoning for Robust Interactive Problem Diagnosis", + "url": "https://arxiv.org/abs/2606.13220", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Fabrizio Marozzo", + "Pietro Liò" + ], + "categories": [ + "cs.AI", + "cs.CE", + "cs.ET", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.13220", + "source": "arxiv", + "source_id": "arxiv:2606.13220", + "pdf_url": "https://arxiv.org/pdf/2606.13220", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.12290", + "title": "Selection Integrity for LLM Graph Memory: An Accumulability Criterion for Information-Flow-Blind Retrieval", + "url": "https://arxiv.org/abs/2606.12290", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Zeming Fei", + "Hongming Fei", + "Xiaoyang Wang", + "Yang yang", + "Prosanta Gope", + "Biplab Sikdar", + "Ying Zhang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.12290", + "source": "arxiv", + "source_id": "arxiv:2606.12290", + "pdf_url": "https://arxiv.org/pdf/2606.12290", + "primary_query": "agent-memory" + }, + { + "id": "2606.12587", + "title": "Strategic Decision Support for AI Agents", + "url": "https://arxiv.org/abs/2606.12587", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Shayan Kiyani", + "Sima Noorani", + "George Pappas", + "Hamed Hassani" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "multi-agent", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.12587", + "source": "arxiv", + "source_id": "arxiv:2606.12587", + "pdf_url": "https://arxiv.org/pdf/2606.12587", + "primary_query": "tool-use" + }, + { + "id": "2606.10651", + "title": "Kwai Keye-VL-2.0 Technical Report", + "url": "https://arxiv.org/abs/2606.10651", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Kwai Keye Team", + "Bin Wen", + "Changyi Liu", + "Chengru Song", + "Chongling Rao", + "Guowang Zhang", + "Han Li", + "Haonan Fan", + "Hengrui Ju", + "Jiankang Chen", + "Jiapeng Chen", + "Jiawei Yuan", + "Kaixuan Yang", + "Kaiyu Jiang", + "Kun Gai", + "Lingzhi Zhou", + "Na Nie", + "Sen Na", + "Tianke Zhang", + "Tingting Gao", + "Xuanyu Zheng", + "Yulong Chen", + "Fan Yang", + "Haixuan Gao", + "Lele Yang", + "Mingqiao Liu", + "Muxi Diao", + "Qi Zhang", + "Qile Su", + "Wei Chen", + "Wentao Hong", + "Xingyu Lu", + "Yancheng Long", + "Yankai Yang", + "Yingxin Li", + "Yiyang Fan", + "Yu Xia", + "Yuzhe Chen", + "Ziliang Lai", + "Chuan Yi", + "Haonan Jia", + "Tianming Liang", + "Weixin Xu", + "Xiaoxiao Ma", + "Yang Tian", + "Yufei Han", + "Feng Han", + "Hang Li", + "Jing Wang", + "Jinghui Jia", + "Junmin Chen", + "Junyu Shi", + "Ruilin Zhang" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.10651", + "source": "arxiv", + "source_id": "arxiv:2606.10651", + "pdf_url": "https://arxiv.org/pdf/2606.10651", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.10956", + "title": "Mind the Gap: Can Frontier LLMs Pass a Standardized Office Proficiency Exam?", + "url": "https://arxiv.org/abs/2606.10956", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Tengchao Lv", + "Dongdong Zhang", + "Jiayu Ding", + "Yilin Jia", + "Yuzhong Zhao", + "Yupan Huang", + "Wenshan Wu", + "Xiangyang Zhou", + "Shaohan Huang", + "Nan Yang", + "Li Dong", + "Lei Cui", + "Furu Wei" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.10956", + "source": "arxiv", + "source_id": "arxiv:2606.10956", + "pdf_url": "https://arxiv.org/pdf/2606.10956", + "primary_query": "planning-agent" + }, + { + "id": "2606.08661", + "title": "Data Agents Under Attack: Vulnerabilities in LLM-Driven Analytical Systems", + "url": "https://arxiv.org/abs/2606.08661", + "published": "2026-06-07", + "updated": "2026-06-07", + "authors": [ + "Kuncan Wang", + "Ziting Wang", + "Peizhuo Lv", + "Haoyang Li", + "Guoliang Li", + "Gao Cong", + "Wei Dong" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.DB" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.08661", + "source": "arxiv", + "source_id": "arxiv:2606.08661", + "pdf_url": "https://arxiv.org/pdf/2606.08661", + "primary_query": "agent-safety" + }, + { + "id": "2606.08539", + "title": "AgentTrust: A Self-Improving Trust Layer for AI-Agent Actions", + "url": "https://arxiv.org/abs/2606.08539", + "published": "2026-06-07", + "updated": "2026-06-07", + "authors": [ + "Chenglin Yang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.08539", + "source": "arxiv", + "source_id": "arxiv:2606.08539", + "pdf_url": "https://arxiv.org/pdf/2606.08539", + "primary_query": "rag-agent" + }, + { + "id": "2606.24896", + "title": "Why Memory Components Fail: Eight Years of License and Sustainability Events in Open-Source Data Infrastructure", + "url": "https://arxiv.org/abs/2606.24896", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Dmitrii Dmitrenko" + ], + "categories": [ + "cs.DL", + "cs.CY" + ], + "topics": [ + "coding-agent", + "memory", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.24896", + "source": "arxiv", + "source_id": "arxiv:2606.24896", + "pdf_url": "https://arxiv.org/pdf/2606.24896", + "primary_query": "agent-memory" + }, + { + "id": "2606.07074", + "title": "SlimSearcher: Training Efficiency-Aware Web Agents via Adaptive Reward Gating", + "url": "https://arxiv.org/abs/2606.07074", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Zequn Xie", + "Junjie Wang", + "Dan Yang", + "Jie Feng", + "Yue Shen", + "Jian Wang", + "Jinjie Gu" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.07074", + "source": "arxiv", + "source_id": "arxiv:2606.07074", + "pdf_url": "https://arxiv.org/pdf/2606.07074", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.07027", + "title": "StainFlow: Entity-Stain Tracking and Evidence Linking for Process Rewards in GUI Agents", + "url": "https://arxiv.org/abs/2606.07027", + "published": "2026-06-05", + "updated": "2026-06-12", + "authors": [ + "Haojie Hao", + "Longkun Hao", + "Yihang Lou", + "Yan Bai", + "Zhenyang Li", + "Zhichao Yang", + "Dongshuo Huang", + "Hongyu Lin", + "Lanqing Hong", + "Jiakai Wang", + "Xianglong Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.07027", + "source": "arxiv", + "source_id": "arxiv:2606.07027", + "pdf_url": "https://arxiv.org/pdf/2606.07027", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.07486", + "title": "OPENPATH: A Supervisor--Specialist Agent System for Personalized, Accessible, and Multi-stop Urban Trip Planning", + "url": "https://arxiv.org/abs/2606.07486", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Ziyang Xiong", + "He Zong", + "Zhiyuan Xue", + "Manxi Wu" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "multi-agent", + "planning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.07486", + "source": "arxiv", + "source_id": "arxiv:2606.07486", + "pdf_url": "https://arxiv.org/pdf/2606.07486", + "primary_query": "planning-agent" + }, + { + "id": "2606.05553", + "title": "ArcANE: Do Role-Playing Language Agents Stay in Character at the Right Time?", + "url": "https://arxiv.org/abs/2606.05553", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Woojung Song", + "Nalim Kim", + "Sangjun Song", + "Chaewon Heo", + "Jongwon Lim", + "Yohan Jo" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.05553", + "source": "arxiv", + "source_id": "arxiv:2606.05553", + "pdf_url": "https://arxiv.org/pdf/2606.05553", + "primary_query": "language-agent" + }, + { + "id": "2606.05901", + "title": "Reducing Hallucinations in Complex Question Answering using Simple Graph-based Retrieval-Augmented Generation (long version)", + "url": "https://arxiv.org/abs/2606.05901", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Christopher J. Wedge", + "Joshua Stutter", + "Danny Dixon", + "Jacek Cała" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.05901", + "source": "arxiv", + "source_id": "arxiv:2606.05901", + "pdf_url": "https://arxiv.org/pdf/2606.05901", + "primary_query": "rag-agent" + }, + { + "id": "2606.05233", + "title": "Domain-Conditioned Safety in Frontier Computer-Using Agents: A 793-Episode Browser Benchmark, a Coding-Domain Cross-Reference, and a Reproducibility Audit of Recent Red-Teaming", + "url": "https://arxiv.org/abs/2606.05233", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Nicholas Saban" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.05233", + "source": "arxiv", + "source_id": "arxiv:2606.05233", + "pdf_url": "https://arxiv.org/pdf/2606.05233", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.09890", + "title": "PreAct-Bench: Benchmarking Predictive Monitoring in LLMs", + "url": "https://arxiv.org/abs/2606.09890", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Hainiu Xu", + "Italo Luis da Silva", + "Jiangnan Ye", + "Yuhao Wang", + "Wei Liu", + "Linyi Yang", + "Jonathan Richard Schwarz", + "Nicola Paoletti", + "Yulan He", + "Hanqi Yan" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.09890", + "source": "arxiv", + "source_id": "arxiv:2606.09890", + "pdf_url": "https://arxiv.org/pdf/2606.09890", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.03054", + "title": "ToolGate: Token-Efficient Pre-Call Control for Tool-Augmented Vision-Language Agents", + "url": "https://arxiv.org/abs/2606.03054", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Anjie Liu", + "Yan Song", + "Zhixun Chen", + "Ziqin Gong", + "Zhongwei Yu", + "Jun Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.03054", + "source": "arxiv", + "source_id": "arxiv:2606.03054", + "pdf_url": "https://arxiv.org/pdf/2606.03054", + "primary_query": "language-agent" + }, + { + "id": "2606.02994", + "title": "Inducing Reasoning Primitives from Agent Traces", + "url": "https://arxiv.org/abs/2606.02994", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Zhihan Lei", + "Jiarui Yan", + "Joshua Momo", + "William W. Cohen" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.02994", + "source": "arxiv", + "source_id": "arxiv:2606.02994", + "pdf_url": "https://arxiv.org/pdf/2606.02994", + "primary_query": "planning-agent" + }, + { + "id": "2606.02754", + "title": "$Ψ$-Bench: Evaluating Persona-Sensitive Influencing in Persuasive Dialogues", + "url": "https://arxiv.org/abs/2606.02754", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Peixuan Han", + "Hongyi Du", + "Jiayu Liu", + "Yihang Sun", + "Yutong Liu", + "Jiaxuan You" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.02754", + "source": "arxiv", + "source_id": "arxiv:2606.02754", + "pdf_url": "https://arxiv.org/pdf/2606.02754", + "primary_query": "language-agent" + }, + { + "id": "2606.02875", + "title": "Handoff Debt: The Rediscovery Cost When Coding Agents Take Over Interrupted Tasks", + "url": "https://arxiv.org/abs/2606.02875", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Dipesh KC", + "Anjila Budathoki" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.02875", + "source": "arxiv", + "source_id": "arxiv:2606.02875", + "pdf_url": "https://arxiv.org/pdf/2606.02875", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.29486", + "title": "PhoneWorld: Scaling Phone-Use Agent Environments", + "url": "https://arxiv.org/abs/2605.29486", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Zhengyang Tang", + "Yuxuan Liu", + "Xin Lai", + "Junyi Li", + "Pengyuan Lyu", + "Jason", + "Yiduo Guo", + "Zhengyao Fang", + "Yang Ding", + "Yi Zhang", + "Weinong Wang", + "Huawen Shen", + "Xingran Zhou", + "Liang Wu", + "Fei Tang", + "Sunqi Fan", + "Shangpin Peng", + "Zheng Ruan", + "Anran Zhang", + "Benyou Wang", + "Rui Yan", + "Ji-Rong Wen", + "Chengquan Zhang", + "Han Hu" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.29486", + "source": "arxiv", + "source_id": "arxiv:2605.29486", + "pdf_url": "https://arxiv.org/pdf/2605.29486", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.28629", + "title": "Mobile-Aptus: Confidence-Driven Proactive and Robust Interaction in MLLM-based Mobile-Using Agents", + "url": "https://arxiv.org/abs/2605.28629", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Zheng Wu", + "Pengzhou Cheng", + "Zongru Wu", + "Yuan Guo", + "Tianjie Ju", + "Aston Zhang", + "Gongshen Liu", + "Zhuosheng Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.28629", + "source": "arxiv", + "source_id": "arxiv:2605.28629", + "pdf_url": "https://arxiv.org/pdf/2605.28629", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.27331", + "title": "Maat: The Agentic Legal Research Assistant for Competition Protection", + "url": "https://arxiv.org/abs/2605.27331", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Basant Mounir", + "Farida Madkour", + "Amira Abdelaziz", + "Asmaa Sami" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.27331", + "source": "arxiv", + "source_id": "arxiv:2605.27331", + "pdf_url": "https://arxiv.org/pdf/2605.27331", + "primary_query": "rag-agent" + }, + { + "id": "2605.25641", + "title": "Iterate Until Retrieved: Factual Nugget Optimization for Discoverable Continual Corrections in Agentic RAG", + "url": "https://arxiv.org/abs/2605.25641", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Moshe Hazoom", + "Gal Patel", + "Alon Talmor", + "Tom Hope" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.25641", + "source": "arxiv", + "source_id": "arxiv:2605.25641", + "pdf_url": "https://arxiv.org/pdf/2605.25641", + "primary_query": "rag-agent" + }, + { + "id": "2605.25480", + "title": "Retrieval as Reasoning: Self-Evolving Agent-Native Retrieval via LLM-Wiki", + "url": "https://arxiv.org/abs/2605.25480", + "published": "2026-05-25", + "updated": "2026-05-26", + "authors": [ + "Haoliang Ming", + "Feifei Li", + "Xiaoqing Wu", + "Wenhui Que" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.25480", + "source": "arxiv", + "source_id": "arxiv:2605.25480", + "pdf_url": "https://arxiv.org/pdf/2605.25480", + "primary_query": "rag-agent" + }, + { + "id": "2605.25002", + "title": "MemMark: State-Evolution Attribution Watermarking for Agent Long-Term Memory Systems", + "url": "https://arxiv.org/abs/2605.25002", + "published": "2026-05-24", + "updated": "2026-05-26", + "authors": [ + "Haobo Zhang", + "Xutao Mao", + "Guangyuan Dong", + "Ziwei Li", + "Xuanbo Su", + "Kaijie Chen", + "Jing Yang", + "Zheng Lin" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "computer-use", + "memory" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.25002", + "source": "arxiv", + "source_id": "arxiv:2605.25002", + "pdf_url": "https://arxiv.org/pdf/2605.25002", + "primary_query": "agent-memory" + }, + { + "id": "2605.20485", + "title": "ZEBRA: Zero-shot Budgeted Resource Allocation for LLM Orchestration", + "url": "https://arxiv.org/abs/2605.20485", + "published": "2026-05-19", + "updated": "2026-05-19", + "authors": [ + "May Hamri", + "Inbal Talgam-Cohen" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.20485", + "source": "arxiv", + "source_id": "arxiv:2605.20485", + "pdf_url": "https://arxiv.org/pdf/2605.20485", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.18597", + "title": "Latent Action Reparameterization for Efficient Agent Inference", + "url": "https://arxiv.org/abs/2605.18597", + "published": "2026-05-18", + "updated": "2026-05-19", + "authors": [ + "Wenhao Huang", + "Qingwen Zeng", + "Qiyue Chen", + "Zijie Guo", + "Yu Sun", + "Cheng Yang", + "Siru Ouyang", + "Jiri Gesi", + "Fang Wu", + "Jiayi Zhang", + "Huaming Chen", + "Bang Liu", + "Xiangru Tang", + "Chenglin Wu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.18597", + "source": "arxiv", + "source_id": "arxiv:2605.18597", + "pdf_url": "https://arxiv.org/pdf/2605.18597", + "primary_query": "planning-agent" + }, + { + "id": "2605.15665", + "title": "PRISM: Prompt Reliability via Iterative Simulation and Monitoring for Enterprise Conversational AI", + "url": "https://arxiv.org/abs/2605.15665", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Keshava Chaitanya", + "Jahnavi Gundakaram" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.15665", + "source": "arxiv", + "source_id": "arxiv:2605.15665", + "pdf_url": "https://arxiv.org/pdf/2605.15665", + "primary_query": "language-agent" + }, + { + "id": "2605.14051", + "title": "SPIN: Structural LLM Planning via Iterative Navigation for Industrial Tasks", + "url": "https://arxiv.org/abs/2605.14051", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Yusuke Ozaki", + "Dhaval Patel" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.14051", + "source": "arxiv", + "source_id": "arxiv:2605.14051", + "pdf_url": "https://arxiv.org/pdf/2605.14051", + "primary_query": "planning-agent" + }, + { + "id": "2605.13037", + "title": "MAP: A Map-then-Act Paradigm for Long-Horizon Interactive Agent Reasoning", + "url": "https://arxiv.org/abs/2605.13037", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Yuxin Liu", + "Ziang Ye", + "Yueqing Sun", + "Mingye Zhu", + "Jinwei Xiao", + "Zhuowen Han", + "Qi GU", + "Xunliang Cai", + "Lei Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.13037", + "source": "arxiv", + "source_id": "arxiv:2605.13037", + "pdf_url": "https://arxiv.org/pdf/2605.13037", + "primary_query": "planning-agent" + }, + { + "id": "2605.11224", + "title": "ABRA: Agent Benchmark for Radiology Applications", + "url": "https://arxiv.org/abs/2605.11224", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Bulat Maksudov", + "Vladislav Kurenkov", + "Kathleen M. Curran", + "Alessandra Mileo" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.11224", + "source": "arxiv", + "source_id": "arxiv:2605.11224", + "pdf_url": "https://arxiv.org/pdf/2605.11224", + "primary_query": "function-calling" + }, + { + "id": "2605.11003", + "title": "The Authorization-Execution Gap Is a Major Safety and Security Problem in Open-World Agents", + "url": "https://arxiv.org/abs/2605.11003", + "published": "2026-05-10", + "updated": "2026-05-10", + "authors": [ + "Baoyuan Wu", + "Qingshan Liu", + "Adel Bibi", + "Irwin King", + "Siwei Lyu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.11003", + "source": "arxiv", + "source_id": "arxiv:2605.11003", + "pdf_url": "https://arxiv.org/pdf/2605.11003", + "primary_query": "agent-safety" + }, + { + "id": "2605.03855", + "title": "Evaluating Generative Models as Interactive Emergent Representations of Human-Like Collaborative Behavior", + "url": "https://arxiv.org/abs/2605.03855", + "published": "2026-05-05", + "updated": "2026-05-06", + "authors": [ + "Shinas Shaji", + "Teena Chakkalayil Hassan", + "Sebastian Houben", + "Alex Mitrevski" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "planning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.03855", + "source": "arxiv", + "source_id": "arxiv:2605.03855", + "pdf_url": "https://arxiv.org/pdf/2605.03855", + "primary_query": "planning-agent" + }, + { + "id": "2605.00380", + "title": "ResRL: Boosting LLM Reasoning via Negative Sample Projection Residual Reinforcement Learning", + "url": "https://arxiv.org/abs/2605.00380", + "published": "2026-05-01", + "updated": "2026-05-08", + "authors": [ + "Zihan Lin", + "Xiaohan Wang", + "Jie Cao", + "Jiajun Chai", + "Li Wang", + "Xiaodong Lu", + "Wei Lin", + "Ran He", + "Guojun Yin" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.00380", + "source": "arxiv", + "source_id": "arxiv:2605.00380", + "pdf_url": "https://arxiv.org/pdf/2605.00380", + "primary_query": "function-calling" + }, + { + "id": "2605.00060", + "title": "TADI: Tool-Augmented Drilling Intelligence via Agentic LLM Orchestration over Heterogeneous Wellsite Data", + "url": "https://arxiv.org/abs/2605.00060", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Rong Lu" + ], + "categories": [ + "cs.AI", + "eess.SY" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.00060", + "source": "arxiv", + "source_id": "arxiv:2605.00060", + "pdf_url": "https://arxiv.org/pdf/2605.00060", + "primary_query": "function-calling" + }, + { + "id": "2604.27132", + "title": "TRUST: A Framework for Decentralized AI Service v.0.1", + "url": "https://arxiv.org/abs/2604.27132", + "published": "2026-04-29", + "updated": "2026-04-29", + "authors": [ + "Yu-Chao Huang", + "Zhen Tan", + "Mohan Zhang", + "Pingzhi Li", + "Zhuo Zhang", + "Tianlong Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27132", + "source": "arxiv", + "source_id": "arxiv:2604.27132", + "pdf_url": "https://arxiv.org/pdf/2604.27132", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.18982", + "title": "SAVOIR: Learning Social Savoir-Faire via Shapley-based Reward Attribution", + "url": "https://arxiv.org/abs/2604.18982", + "published": "2026-04-21", + "updated": "2026-04-21", + "authors": [ + "Xiachong Feng", + "Yi Jiang", + "Xiaocheng Feng", + "Deyi Yin", + "Libo Qin", + "Yangfan Ye", + "Lei Huang", + "Weitao Ma", + "Yuxuan Gu", + "Chonghan Qin", + "Bing Qin", + "Lingpeng Kong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.18982", + "source": "arxiv", + "source_id": "arxiv:2604.18982", + "pdf_url": "https://arxiv.org/pdf/2604.18982", + "primary_query": "language-agent" + }, + { + "id": "2604.03070", + "title": "How Your Credentials Are Leaked by LLM Agent Skills: An Empirical Study", + "url": "https://arxiv.org/abs/2604.03070", + "published": "2026-04-03", + "updated": "2026-06-19", + "authors": [ + "Zhihao Chen", + "Ying Zhang", + "Yi Liu", + "Gelei Deng", + "Yuekang Li", + "Yanjun Zhang", + "Jianting Ning", + "Leo Yu Zhang", + "Lei Ma", + "Zhiqiang Li" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.03070", + "source": "arxiv", + "source_id": "arxiv:2604.03070", + "pdf_url": "https://arxiv.org/pdf/2604.03070", + "primary_query": "agent-safety" + }, + { + "id": "2604.00830", + "title": "Learning to Learn-at-Test-Time: Language Agents with Learnable Adaptation Policies", + "url": "https://arxiv.org/abs/2604.00830", + "published": "2026-04-01", + "updated": "2026-04-02", + "authors": [ + "Zhanzhi Lou", + "Hui Chen", + "Yibo Li", + "Qian Wang", + "Bryan Hooi" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.00830", + "source": "arxiv", + "source_id": "arxiv:2604.00830", + "pdf_url": "https://arxiv.org/pdf/2604.00830", + "primary_query": "language-agent" + }, + { + "id": "2604.00992", + "title": "Tube-Based Safety for Anticipative Tracking in Multi-Agent Systems", + "url": "https://arxiv.org/abs/2604.00992", + "published": "2026-04-01", + "updated": "2026-04-01", + "authors": [ + "Armel Koulong", + "Ali Pakniyat" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-safety", + "multi-agent", + "world-model" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.00992", + "source": "arxiv", + "source_id": "arxiv:2604.00992", + "pdf_url": "https://arxiv.org/pdf/2604.00992", + "primary_query": "agent-safety" + }, + { + "id": "2603.29560", + "title": "Distributed Predictive Control Barrier Functions: Towards Scalable Safety Certification in Modular Multi-Agent Systems", + "url": "https://arxiv.org/abs/2603.29560", + "published": "2026-03-31", + "updated": "2026-03-31", + "authors": [ + "Jonas Ohnemus", + "Alexandre Didier", + "Ahmed Aboudonia", + "Andrea Carron", + "Melanie N. Zeilinger" + ], + "categories": [ + "eess.SY", + "cs.RO", + "math.OC" + ], + "topics": [ + "agent-safety", + "multi-agent", + "world-model" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.29560", + "source": "arxiv", + "source_id": "arxiv:2603.29560", + "pdf_url": "https://arxiv.org/pdf/2603.29560", + "primary_query": "agent-safety" + }, + { + "id": "2603.12644", + "title": "Uncovering Security Threats and Architecting Defenses in Autonomous Agents: A Case Study of OpenClaw", + "url": "https://arxiv.org/abs/2603.12644", + "published": "2026-03-13", + "updated": "2026-03-13", + "authors": [ + "Zonghao Ying", + "Xiao Yang", + "Siyang Wu", + "Yumeng Song", + "Yang Qu", + "Hainan Li", + "Tianlin Li", + "Jiakai Wang", + "Aishan Liu", + "Xianglong Liu" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.12644", + "source": "arxiv", + "source_id": "arxiv:2603.12644", + "pdf_url": "https://arxiv.org/pdf/2603.12644", + "primary_query": "agent-safety" + }, + { + "id": "2603.10213", + "title": "Sabiá-4 Technical Report", + "url": "https://arxiv.org/abs/2603.10213", + "published": "2026-03-10", + "updated": "2026-03-10", + "authors": [ + "Thiago Laitz", + "Thales Sales Almeida", + "Hugo Abonizio", + "Roseval Malaquias Junior", + "Giovana Kerche Bonás", + "Marcos Piau", + "Celio Larcher", + "Ramon Pires", + "Rodrigo Nogueira" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.10213", + "source": "arxiv", + "source_id": "arxiv:2603.10213", + "pdf_url": "https://arxiv.org/pdf/2603.10213", + "primary_query": "function-calling" + }, + { + "id": "2603.08316", + "title": "SlowBA: An efficiency backdoor attack towards VLM-based GUI agents", + "url": "https://arxiv.org/abs/2603.08316", + "published": "2026-03-09", + "updated": "2026-07-01", + "authors": [ + "Junxian Li", + "Tu Lan", + "Haozhen Tan", + "Yan Meng", + "Haojin Zhu" + ], + "categories": [ + "cs.CR", + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-safety", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.08316", + "source": "arxiv", + "source_id": "arxiv:2603.08316", + "pdf_url": "https://arxiv.org/pdf/2603.08316", + "primary_query": "agent-safety" + }, + { + "id": "2602.23876", + "title": "RF-Agent: Automated Reward Function Design via Language Agent Tree Search", + "url": "https://arxiv.org/abs/2602.23876", + "published": "2026-02-27", + "updated": "2026-02-27", + "authors": [ + "Ning Gao", + "Xiuhui Zhang", + "Xingyu Jiang", + "Mukang You", + "Mohan Zhang", + "Yue Deng" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "rag", + "reasoning" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.23876", + "source": "arxiv", + "source_id": "arxiv:2602.23876", + "pdf_url": "https://arxiv.org/pdf/2602.23876", + "primary_query": "language-agent" + }, + { + "id": "2602.20156", + "title": "Skill-Inject: Measuring Agent Vulnerability to Skill File Attacks", + "url": "https://arxiv.org/abs/2602.20156", + "published": "2026-02-23", + "updated": "2026-02-25", + "authors": [ + "David Schmotz", + "Luca Beurer-Kellner", + "Sahar Abdelnabi", + "Maksym Andriushchenko" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.20156", + "source": "arxiv", + "source_id": "arxiv:2602.20156", + "pdf_url": "https://arxiv.org/pdf/2602.20156", + "primary_query": "agent-safety" + }, + { + "id": "2602.17875", + "title": "MultiVer: Zero-Shot Multi-Agent Vulnerability Detection", + "url": "https://arxiv.org/abs/2602.17875", + "published": "2026-02-19", + "updated": "2026-02-19", + "authors": [ + "Shreshth Rajan" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.17875", + "source": "arxiv", + "source_id": "arxiv:2602.17875", + "pdf_url": "https://arxiv.org/pdf/2602.17875", + "primary_query": "agent-safety" + }, + { + "id": "2512.00332", + "title": "Assertion-Conditioned Compliance: A Provenance-Aware Vulnerability in Multi-Turn Tool-Calling Agents", + "url": "https://arxiv.org/abs/2512.00332", + "published": "2025-11-29", + "updated": "2026-01-21", + "authors": [ + "Daud Waqas", + "Aaryamaan Golthi", + "Erika Hayashida", + "Huanzhi Mao" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.00332", + "source": "arxiv", + "source_id": "arxiv:2512.00332", + "pdf_url": "https://arxiv.org/pdf/2512.00332", + "primary_query": "function-calling" + }, + { + "id": "2510.24284", + "title": "MCP-Flow: Facilitating LLM Agents to Master Real-World, Diverse and Scaling MCP Tools", + "url": "https://arxiv.org/abs/2510.24284", + "published": "2025-10-28", + "updated": "2026-04-16", + "authors": [ + "Wenhao Wang", + "Peizhi Niu", + "Zhao Xu", + "Zhaoyu Chen", + "Jian Du", + "Yaxin Du", + "Xianghe Pang", + "Keduan Huang", + "Yanfeng Wang", + "Qiang Yan", + "Siheng Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.24284", + "source": "arxiv", + "source_id": "arxiv:2510.24284", + "pdf_url": "https://arxiv.org/pdf/2510.24284", + "primary_query": "function-calling" + }, + { + "id": "2509.18420", + "title": "Instruction-Following Evaluation in Function Calling for Large Language Models", + "url": "https://arxiv.org/abs/2509.18420", + "published": "2025-09-22", + "updated": "2025-09-22", + "authors": [ + "Nikolai Skripko" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.18420", + "source": "arxiv", + "source_id": "arxiv:2509.18420", + "pdf_url": "https://arxiv.org/pdf/2509.18420", + "primary_query": "function-calling" + }, + { + "id": "2509.18169", + "title": "PiERN: Token-Level Routing for Integrating High-Precision Computation and Reasoning", + "url": "https://arxiv.org/abs/2509.18169", + "published": "2025-09-17", + "updated": "2026-04-20", + "authors": [ + "Hengbo Xiao", + "Jingyuan Fan", + "Xin Tong", + "Jingzhao Zhang", + "Chao Lu", + "Guannan He" + ], + "categories": [ + "cs.LG", + "cs.CE", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.18169", + "source": "arxiv", + "source_id": "arxiv:2509.18169", + "pdf_url": "https://arxiv.org/pdf/2509.18169", + "primary_query": "function-calling" + }, + { + "id": "2508.20637", + "title": "GDS Agent for Graph Algorithmic Reasoning", + "url": "https://arxiv.org/abs/2508.20637", + "published": "2025-08-28", + "updated": "2025-11-05", + "authors": [ + "Borun Shi", + "Ioannis Panagiotas" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 12, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.20637", + "source": "arxiv", + "source_id": "arxiv:2508.20637", + "pdf_url": "https://arxiv.org/pdf/2508.20637", + "primary_query": "function-calling" + }, + { + "id": "2607.06184", + "title": "What Resolve Rate Hides: Trajectory Structure Diagnostics for Coding Agents", + "url": "https://arxiv.org/abs/2607.06184", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Rui Shu", + "Chun Yong Chong", + "Xin Zhou", + "Yun Peng", + "Zihan Wu", + "Xu Han", + "Zeyang Zhuang", + "Guowen Yuan", + "Yuan Wang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "coding-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.06184", + "source": "arxiv", + "source_id": "arxiv:2607.06184", + "pdf_url": "https://arxiv.org/pdf/2607.06184", + "primary_query": "coding-agent" + }, + { + "id": "2607.06065", + "title": "SWE-Review: Closing the Loop on Issue Resolution with Agentic Code Review", + "url": "https://arxiv.org/abs/2607.06065", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Ruoyu Wang", + "Jierun Chen", + "Shaowei Wang", + "Chaofan Tao", + "Sidi Yang", + "Yuxin Jiang", + "Kim-Hui Yap", + "Lifeng Shang", + "Xiaohui Li", + "Haoli Bai" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.06065", + "source": "arxiv", + "source_id": "arxiv:2607.06065", + "pdf_url": "https://arxiv.org/pdf/2607.06065", + "primary_query": "coding-agent" + }, + { + "id": "2607.05785", + "title": "Can Large Language Models Generate Observability-Aware Code?", + "url": "https://arxiv.org/abs/2607.05785", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Yongliang Tao", + "Hongyu Zhang", + "Pengfei Gao", + "Minghua Ma", + "Zhiyu Fan", + "Yu Kang", + "Jue Zhang", + "Si Qin", + "Liqun Li", + "Qingwei Lin", + "Saravan Rajmohan" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.05785", + "source": "arxiv", + "source_id": "arxiv:2607.05785", + "pdf_url": "https://arxiv.org/pdf/2607.05785", + "primary_query": "coding-agent" + }, + { + "id": "2607.05682", + "title": "FirstResearch: Auditable Question Formation for LLM Scientific Discovery Agents", + "url": "https://arxiv.org/abs/2607.05682", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Yufeng Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "planning", + "rag" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.05682", + "source": "arxiv", + "source_id": "arxiv:2607.05682", + "pdf_url": "https://arxiv.org/pdf/2607.05682", + "primary_query": "llm-agent" + }, + { + "id": "2607.04576", + "title": "Progressive Disclosure for LLM-Maintained Wiki Knowledge Bases: a Preregistered Ablation", + "url": "https://arxiv.org/abs/2607.04576", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Theodore O. Cochran" + ], + "categories": [ + "cs.CL", + "cs.CY", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.04576", + "source": "arxiv", + "source_id": "arxiv:2607.04576", + "pdf_url": "https://arxiv.org/pdf/2607.04576", + "primary_query": "llm-agent" + }, + { + "id": "2607.04558", + "title": "EEG-SpikeAgent: Agentic Closed-Loop Program Synthesis for Automated EEG Spike Detection", + "url": "https://arxiv.org/abs/2607.04558", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Sonali Santhosh", + "Kelly Shuhong Yu", + "Eugene Chang", + "Jonathan Kim", + "Kie Shidara", + "Danilo Bernardo" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04558", + "source": "arxiv", + "source_id": "arxiv:2607.04558", + "pdf_url": "https://arxiv.org/pdf/2607.04558", + "primary_query": "llm-agent" + }, + { + "id": "2607.05593", + "title": "Collective Cognition in Hybrid Groups: A Network Science Synthesis", + "url": "https://arxiv.org/abs/2607.05593", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Babak Hemmatian", + "Razan Baltaji", + "Lav R. Varshney" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "memory", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai", + "ai-agent" + ], + "arxiv_id": "2607.05593", + "source": "arxiv", + "source_id": "arxiv:2607.05593", + "pdf_url": "https://arxiv.org/pdf/2607.05593", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04948", + "title": "Using Process Mining to Generate AI Agents from Software Engineering Process Records", + "url": "https://arxiv.org/abs/2607.04948", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Saimir Bala", + "Fabiana Fournier", + "Lior Limonad", + "Andreas Metzger" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "coding-agent", + "multi-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.04948", + "source": "arxiv", + "source_id": "arxiv:2607.04948", + "pdf_url": "https://arxiv.org/pdf/2607.04948", + "primary_query": "ai-agent" + }, + { + "id": "2607.05462", + "title": "Evaluating calibrated refusal and safe usefulness in dual-use biology settings", + "url": "https://arxiv.org/abs/2607.05462", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Edwin H. Wintermute", + "Harmon Bhasin", + "Christina M. Agapakis", + "Dianzhuo Wang", + "Evan Seeyave", + "Arjun Banerjee", + "Daniel Fulop", + "Matthew C. Watson", + "Adam J. Meyer", + "Sandrine Boissel", + "Jens H. Kuhn", + "Rishi Jain", + "Noah D. Taylor", + "Helena Shomar", + "Patrick M. Boyle", + "Kenny Workman" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.05462", + "source": "arxiv", + "source_id": "arxiv:2607.05462", + "pdf_url": "https://arxiv.org/pdf/2607.05462", + "primary_query": "ai-agent" + }, + { + "id": "2607.05483", + "title": "PatchOptic for Shared-State LLM Workflows with Projected Views and Verified Structured Updates", + "url": "https://arxiv.org/abs/2607.05483", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Zhaoyu Bai", + "Jiaqi Cai" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.LO", + "cs.MA", + "cs.PL" + ], + "topics": [ + "agent-evaluation", + "rag", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai", + "rag-agent" + ], + "arxiv_id": "2607.05483", + "source": "arxiv", + "source_id": "arxiv:2607.05483", + "pdf_url": "https://arxiv.org/pdf/2607.05483", + "primary_query": "agentic-ai" + }, + { + "id": "2607.05139", + "title": "On the risk of coding before testing: An empirical study on LLM-based test generation workflow", + "url": "https://arxiv.org/abs/2607.05139", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Michael Konstantinou", + "Florian Tambon", + "Mike Papadakis" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.05139", + "source": "arxiv", + "source_id": "arxiv:2607.05139", + "pdf_url": "https://arxiv.org/pdf/2607.05139", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04096", + "title": "Forethought: Verifiable Reasoning from Neurosymbolic Primitive Programming", + "url": "https://arxiv.org/abs/2607.04096", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Vishvesh Bhat", + "Jay Vaghasiya", + "Emmanuel Anaya Gonzalez" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04096", + "source": "arxiv", + "source_id": "arxiv:2607.04096", + "pdf_url": "https://arxiv.org/pdf/2607.04096", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04425", + "title": "UI-MOPD: Multi-Platform On-Policy Distillation for Continual GUI Agent Learning", + "url": "https://arxiv.org/abs/2607.04425", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Niu Lian", + "Alan Chen", + "Zhehao Yu", + "Chengzhen Duan", + "Fazhan Liu", + "Hui Liu", + "Pei Fu", + "Jian Luan", + "Yaowei Wang", + "Shu-Tao Xia", + "Jinpeng Wang" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.CV", + "cs.LG", + "cs.MM" + ], + "topics": [ + "computer-use", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.04425", + "source": "arxiv", + "source_id": "arxiv:2607.04425", + "pdf_url": "https://arxiv.org/pdf/2607.04425", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.03651", + "title": "LLM-Guided Transportation Hub Capacity Planning with Textual Business Inputs", + "url": "https://arxiv.org/abs/2607.03651", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Xiaoyue Liu", + "Zheng Dong" + ], + "categories": [ + "cs.LG", + "math.OC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.03651", + "source": "arxiv", + "source_id": "arxiv:2607.03651", + "pdf_url": "https://arxiv.org/pdf/2607.03651", + "primary_query": "llm-agent" + }, + { + "id": "2607.02975", + "title": "Evaluating Generative Agents with Actions Grounded in Socially Distributed Task Environments using Incognita", + "url": "https://arxiv.org/abs/2607.02975", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Dan C. Hsu", + "Luke Lu" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2607.02975", + "source": "arxiv", + "source_id": "arxiv:2607.02975", + "pdf_url": "https://arxiv.org/pdf/2607.02975", + "primary_query": "language-agent" + }, + { + "id": "2607.03025", + "title": "Human-Centric Reflective Architecture for Human-AI Collaborative Decision-Making", + "url": "https://arxiv.org/abs/2607.03025", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Andreas Kouridakis", + "Dimitrios Patiniotis Spyropoulos", + "George Vouros" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.03025", + "source": "arxiv", + "source_id": "arxiv:2607.03025", + "pdf_url": "https://arxiv.org/pdf/2607.03025", + "primary_query": "ai-agent" + }, + { + "id": "2607.03386", + "title": "When Aggregate Alignment Misleads: Auditing Policy Repair Without Per-State Expert Actions", + "url": "https://arxiv.org/abs/2607.03386", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Peiying Zhu", + "Sidi Chang" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.03386", + "source": "arxiv", + "source_id": "arxiv:2607.03386", + "pdf_url": "https://arxiv.org/pdf/2607.03386", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03332", + "title": "AtomicCommitBench: Can Coding Agents Reconstruct Commit Histories from Squashed Patches?", + "url": "https://arxiv.org/abs/2607.03332", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Zhihao Lin", + "Mingyi Zhou", + "Li Li" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03332", + "source": "arxiv", + "source_id": "arxiv:2607.03332", + "pdf_url": "https://arxiv.org/pdf/2607.03332", + "primary_query": "coding-agent" + }, + { + "id": "2607.01810", + "title": "Decoupling Code Complexity from Newcomer Participation: A Causal Study of AI Coding Agent Adoption in OSS", + "url": "https://arxiv.org/abs/2607.01810", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Weiwei Xu", + "Xuanning Cui", + "Hengzhi Ye", + "Minghui Zhou" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.01810", + "source": "arxiv", + "source_id": "arxiv:2607.01810", + "pdf_url": "https://arxiv.org/pdf/2607.01810", + "primary_query": "coding-agent" + }, + { + "id": "2607.01760", + "title": "Refploit: Facilitating Exploit Construction via Code-Agent Trajectory Repair", + "url": "https://arxiv.org/abs/2607.01760", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Zirui Chen", + "Zhipeng Xue", + "Jiayuan Zhou", + "Xing Hu", + "Xin Xia", + "Xiaohu Yang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.01760", + "source": "arxiv", + "source_id": "arxiv:2607.01760", + "pdf_url": "https://arxiv.org/pdf/2607.01760", + "primary_query": "coding-agent" + }, + { + "id": "2607.01418", + "title": "Adoption and Impact of Command-Line AI Coding Agents: A Study of Microsoft's Early 2026 Rollout of Claude Code and GitHub Copilot CLI", + "url": "https://arxiv.org/abs/2607.01418", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Emerson Murphy-Hill", + "Jenna Butler", + "Alexandra Savelieva" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.HC" + ], + "topics": [ + "coding-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.01418", + "source": "arxiv", + "source_id": "arxiv:2607.01418", + "pdf_url": "https://arxiv.org/pdf/2607.01418", + "primary_query": "coding-agent" + }, + { + "id": "2607.01087", + "title": "Cheap Code, Costly Judgment: A Case Study on Governable Agentic Software Engineering", + "url": "https://arxiv.org/abs/2607.01087", + "published": "2026-07-01", + "updated": "2026-07-04", + "authors": [ + "James C. Davis", + "Paschal C. Amusuo", + "Tanmay Singla", + "Berk Çakar", + "Kirsten A. Davis" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.01087", + "source": "arxiv", + "source_id": "arxiv:2607.01087", + "pdf_url": "https://arxiv.org/pdf/2607.01087", + "primary_query": "coding-agent" + }, + { + "id": "2607.00895", + "title": "Beyond Document Grounding: Span-Level Hallucination Detection over Code, Tool Output, and Documents", + "url": "https://arxiv.org/abs/2607.00895", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Ádám Kovács", + "Bowei He", + "Xue Liu", + "István Boros", + "Szilveszter Tóth", + "Gábor Recski" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent", + "rag-agent" + ], + "arxiv_id": "2607.00895", + "source": "arxiv", + "source_id": "arxiv:2607.00895", + "pdf_url": "https://arxiv.org/pdf/2607.00895", + "primary_query": "coding-agent" + }, + { + "id": "2607.01063", + "title": "AutoRestTest at the SBFT 2026 Tool Competition", + "url": "https://arxiv.org/abs/2607.01063", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Tyler Stennett", + "Myeongsoo Kim", + "Saurabh Sinha", + "Alessandro Orso" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.01063", + "source": "arxiv", + "source_id": "arxiv:2607.01063", + "pdf_url": "https://arxiv.org/pdf/2607.01063", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31371", + "title": "Calibrating the Evaluator: Does Probability Calibration Mitigate Preference Coupling in LLM Agent Feedback Loops?", + "url": "https://arxiv.org/abs/2606.31371", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Zewen Liu" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.31371", + "source": "arxiv", + "source_id": "arxiv:2606.31371", + "pdf_url": "https://arxiv.org/pdf/2606.31371", + "primary_query": "llm-agent" + }, + { + "id": "2606.31182", + "title": "AI-Assisted Discovery of Convex Relaxations via Dual Agents", + "url": "https://arxiv.org/abs/2606.31182", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Sungyoon Kim", + "Mert Pilanci" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "coding-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2606.31182", + "source": "arxiv", + "source_id": "arxiv:2606.31182", + "pdf_url": "https://arxiv.org/pdf/2606.31182", + "primary_query": "coding-agent" + }, + { + "id": "2606.31935", + "title": "Delegation Rights: Property, Agency, and Investment Incentives in the Age of AI Agents", + "url": "https://arxiv.org/abs/2606.31935", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yukun Zhang", + "Kemu Xu" + ], + "categories": [ + "econ.EM" + ], + "topics": [ + "agent-safety", + "tool-use", + "world-model" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.31935", + "source": "arxiv", + "source_id": "arxiv:2606.31935", + "pdf_url": "https://arxiv.org/pdf/2606.31935", + "primary_query": "ai-agent" + }, + { + "id": "2606.31041", + "title": "A Semantic-Layer-Mediated Agent for Natural Language to SQL over Heterogeneous Enterprise Databases", + "url": "https://arxiv.org/abs/2606.31041", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Ha Jeong Kim", + "Saksonita Khoeurn", + "Ye Ji Yoon" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.31041", + "source": "arxiv", + "source_id": "arxiv:2606.31041", + "pdf_url": "https://arxiv.org/pdf/2606.31041", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31270", + "title": "Learning from Failure: Inference-Time Self-Improvement for Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.31270", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Xueqiao Sun", + "Xiaohan Wang", + "Ludwig Schmidt", + "Serena Yeung-Levy", + "Yuhui Zhang" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.CL", + "cs.CY", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.31270", + "source": "arxiv", + "source_id": "arxiv:2606.31270", + "pdf_url": "https://arxiv.org/pdf/2606.31270", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.31154", + "title": "PPT-Eval: A Benchmark for Computer-Use Agents on PowerPoint Tasks", + "url": "https://arxiv.org/abs/2606.31154", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Apurva Gandhi", + "Vishwas Suryanarayanan", + "Raja Hasnain Anwar", + "Firoz Shaik", + "Shubhang Desai", + "Thong Q. Nguyen", + "Muhammad Taqi Raza", + "Vishal Chowdhary", + "Graham Neubig" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.31154", + "source": "arxiv", + "source_id": "arxiv:2606.31154", + "pdf_url": "https://arxiv.org/pdf/2606.31154", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.31273", + "title": "The Calibration Turn in AI-Assisted Research: A Conceptual and Methodological Framework for Evidence-Licensed Claims", + "url": "https://arxiv.org/abs/2606.31273", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Hongmin Li" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31273", + "source": "arxiv", + "source_id": "arxiv:2606.31273", + "pdf_url": "https://arxiv.org/pdf/2606.31273", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29717", + "title": "Optimizing Expert-Designed Crystal Graph Networks for Band-Gap Prediction with an Autonomous LLM Research Loop", + "url": "https://arxiv.org/abs/2606.29717", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Chenmu Zhang", + "Boris I. Yakobson" + ], + "categories": [ + "cond-mat.mtrl-sci", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm", + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2606.29717", + "source": "arxiv", + "source_id": "arxiv:2606.29717", + "pdf_url": "https://arxiv.org/pdf/2606.29717", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.29705", + "title": "GUICrafter: Weakly-Supervised GUI Agent Leveraging Massive Unannotated Screenshots", + "url": "https://arxiv.org/abs/2606.29705", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Sunqi Fan", + "Lingshan Chen", + "Runqi Yin", + "Qingle Liu", + "Yongming Rao", + "Meng-Hao Guo", + "Shi-Min Hu" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CV" + ], + "topics": [ + "computer-use", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.29705", + "source": "arxiv", + "source_id": "arxiv:2606.29705", + "pdf_url": "https://arxiv.org/pdf/2606.29705", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.30571", + "title": "Attractor States Emerge in Multi-Turn LLM Conversations", + "url": "https://arxiv.org/abs/2606.30571", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Ting-Wen Ko", + "Jonas Geiping" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm", + "multi-agent-llm" + ], + "arxiv_id": "2606.30571", + "source": "arxiv", + "source_id": "arxiv:2606.30571", + "pdf_url": "https://arxiv.org/pdf/2606.30571", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.29981", + "title": "Hephaestus: Toward a Cybersecurity AI Scientist", + "url": "https://arxiv.org/abs/2606.29981", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Jiaqi Li", + "Yang Zhao", + "Wen Lu", + "Lvyang Zhang", + "Lidong Zhai" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.29981", + "source": "arxiv", + "source_id": "arxiv:2606.29981", + "pdf_url": "https://arxiv.org/pdf/2606.29981", + "primary_query": "agent-safety" + }, + { + "id": "2606.29406", + "title": "Adaptive AI Delegation under Uncertainty: A Bayesian Governance Policy for Sequential Decision Authority", + "url": "https://arxiv.org/abs/2606.29406", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Matthew Francis Dixon" + ], + "categories": [ + "q-fin.RM", + "math.OC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.29406", + "source": "arxiv", + "source_id": "arxiv:2606.29406", + "pdf_url": "https://arxiv.org/pdf/2606.29406", + "primary_query": "agentic-ai" + }, + { + "id": "2606.29648", + "title": "Hybrid Retriever Evolution for Multimodal Document Reasoning Agents", + "url": "https://arxiv.org/abs/2606.29648", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Bohan Yao", + "Shruthan Radhakrishna", + "Vikas Yadav" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.29648", + "source": "arxiv", + "source_id": "arxiv:2606.29648", + "pdf_url": "https://arxiv.org/pdf/2606.29648", + "primary_query": "tool-use" + }, + { + "id": "2606.29472", + "title": "Agent-Computer Observation Interfaces Enable Dynamic Computer Use", + "url": "https://arxiv.org/abs/2606.29472", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Bojie Li", + "Noah Shi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent", + "web-gui-agent" + ], + "arxiv_id": "2606.29472", + "source": "arxiv", + "source_id": "arxiv:2606.29472", + "pdf_url": "https://arxiv.org/pdf/2606.29472", + "primary_query": "coding-agent" + }, + { + "id": "2606.29605", + "title": "How much of an LLM-generated clinical corpus is actually new? A production-scale measurement of content redundancy for provenance classification", + "url": "https://arxiv.org/abs/2606.29605", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Ali H. Lazem", + "William J. Teahan" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.29605", + "source": "arxiv", + "source_id": "arxiv:2606.29605", + "pdf_url": "https://arxiv.org/pdf/2606.29605", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.27650", + "title": "GenWorld: Empirically Grounded Urban Simulation Infrastructure for Scalable LLM-Agent Studies", + "url": "https://arxiv.org/abs/2606.27650", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Gen Li", + "Jieyuan Lan", + "Pengcheng Xu", + "Zongyuan Wu", + "Masaki Ogura", + "Tao Feng" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "computer-use", + "planning", + "rag", + "world-model" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.27650", + "source": "arxiv", + "source_id": "arxiv:2606.27650", + "pdf_url": "https://arxiv.org/pdf/2606.27650", + "primary_query": "llm-agent" + }, + { + "id": "2606.28235", + "title": "Govern the Repository, Not the Agent: Measuring Ecosystem-Level Risk in AI-Native Software", + "url": "https://arxiv.org/abs/2606.28235", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Daniel Russo" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2606.28235", + "source": "arxiv", + "source_id": "arxiv:2606.28235", + "pdf_url": "https://arxiv.org/pdf/2606.28235", + "primary_query": "agentic-ai" + }, + { + "id": "2606.27909", + "title": "Triadic Werewolf: A Jester Role for Multi-Hop Theory of Mind in LLMs", + "url": "https://arxiv.org/abs/2606.27909", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Avni Mittal" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.GT", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.27909", + "source": "arxiv", + "source_id": "arxiv:2606.27909", + "pdf_url": "https://arxiv.org/pdf/2606.27909", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28002", + "title": "Dialogue to Detection: A Multimodal Hybrid NLP Pipeline for Insurance Fraud Detection", + "url": "https://arxiv.org/abs/2606.28002", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Muhammad Shakeel Akram", + "Amal Htait", + "Abdul Hamid Sadka", + "Emma Meisingseth", + "Karishma Jaitly" + ], + "categories": [ + "cs.CL", + "cs.AI", + "eess.AS" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.28002", + "source": "arxiv", + "source_id": "arxiv:2606.28002", + "pdf_url": "https://arxiv.org/pdf/2606.28002", + "primary_query": "rag-agent" + }, + { + "id": "2606.26722", + "title": "Socratic agents for autonomous scientific discovery in high-dimensional physical systems", + "url": "https://arxiv.org/abs/2606.26722", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Xianrui Zeng", + "Pengfei Liu", + "Yirui Zang", + "Yang Shen", + "Fei Yu", + "Chunlei Yu", + "Minghao Liu", + "Yang Du" + ], + "categories": [ + "cs.AI", + "physics.optics" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.26722", + "source": "arxiv", + "source_id": "arxiv:2606.26722", + "pdf_url": "https://arxiv.org/pdf/2606.26722", + "primary_query": "agentic-ai" + }, + { + "id": "2606.27595", + "title": "Ko-WideSearch: A Korean Breadth-Search Benchmark for Exhaustive Set Enumeration by Web Agents", + "url": "https://arxiv.org/abs/2606.27595", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Minbyul Jeong" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation", + "web-gui-agent" + ], + "arxiv_id": "2606.27595", + "source": "arxiv", + "source_id": "arxiv:2606.27595", + "pdf_url": "https://arxiv.org/pdf/2606.27595", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.26474", + "title": "Localizing RL-Induced Tool Use to a Single Crosscoder Feature", + "url": "https://arxiv.org/abs/2606.26474", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Andrii Shportko", + "Shubham Bhokare", + "Ahmed Zeyad A Alzahrani", + "Bowen Cheng", + "Gustavo Mercier", + "Jessica Hullman" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.26474", + "source": "arxiv", + "source_id": "arxiv:2606.26474", + "pdf_url": "https://arxiv.org/pdf/2606.26474", + "primary_query": "tool-use" + }, + { + "id": "2606.27122", + "title": "Mostly Automatic Translation of Language Interpreters from C to Safe Rust", + "url": "https://arxiv.org/abs/2606.27122", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Bo Wang", + "Brandon Paulsen", + "Joey Dodds", + "Daniel Kroening", + "Umang Mathur", + "Prateek Saxena" + ], + "categories": [ + "cs.PL", + "cs.MA", + "cs.SE" + ], + "topics": [ + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.27122", + "source": "arxiv", + "source_id": "arxiv:2606.27122", + "pdf_url": "https://arxiv.org/pdf/2606.27122", + "primary_query": "coding-agent" + }, + { + "id": "2606.26979", + "title": "How Much Static Structure Do Code Agents Need? A Study of Deterministic Anchoring", + "url": "https://arxiv.org/abs/2606.26979", + "published": "2026-06-25", + "updated": "2026-07-02", + "authors": [ + "Zhihao Lin", + "Mingyi Zhou", + "Yizhuo Yang", + "Li Li" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "coding-agent", + "computer-use", + "embodied-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.26979", + "source": "arxiv", + "source_id": "arxiv:2606.26979", + "pdf_url": "https://arxiv.org/pdf/2606.26979", + "primary_query": "coding-agent" + }, + { + "id": "2606.26298", + "title": "Governing Actions, Not Agents: Institutional Attestation as a Governance Model for Autonomous AI Systems", + "url": "https://arxiv.org/abs/2606.26298", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Jakob Salfeld-Nebgen" + ], + "categories": [ + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.26298", + "source": "arxiv", + "source_id": "arxiv:2606.26298", + "pdf_url": "https://arxiv.org/pdf/2606.26298", + "primary_query": "ai-agent" + }, + { + "id": "2606.26027", + "title": "Why Multi-Step Tool-Use Reinforcement Learning Collapses and How Supervisory Signals Fix It", + "url": "https://arxiv.org/abs/2606.26027", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yupu Hao", + "Zhuoran Jin", + "Huanxuan Liao", + "Kang Liu", + "Jun Zhao" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.26027", + "source": "arxiv", + "source_id": "arxiv:2606.26027", + "pdf_url": "https://arxiv.org/pdf/2606.26027", + "primary_query": "tool-use" + }, + { + "id": "2606.25257", + "title": "How Do Developers Maintain and Evolve Their Agents' Instructions? An Empirical Study", + "url": "https://arxiv.org/abs/2606.25257", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Gianmario Voria", + "Alfonso Cannavale", + "Andrea De Lucia", + "Yutaro Kashiwa", + "Gemma Catolino", + "Fabio Palomba" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.25257", + "source": "arxiv", + "source_id": "arxiv:2606.25257", + "pdf_url": "https://arxiv.org/pdf/2606.25257", + "primary_query": "coding-agent" + }, + { + "id": "2606.24429", + "title": "Detecting AI Coding Agents in Open Source: A Validated Multi-Method Census of 180 Million Repositories", + "url": "https://arxiv.org/abs/2606.24429", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Arsham Khosravani", + "Audris Mockus" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.24429", + "source": "arxiv", + "source_id": "arxiv:2606.24429", + "pdf_url": "https://arxiv.org/pdf/2606.24429", + "primary_query": "coding-agent" + }, + { + "id": "2606.24965", + "title": "Project Auto-World: Towards Automated Benchmarking of Neural Relational Reasoners", + "url": "https://arxiv.org/abs/2606.24965", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Anirban Das", + "Joanne Boisson", + "Irtaza Khalid", + "Sumita Garai", + "Steven Schockaert" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.24965", + "source": "arxiv", + "source_id": "arxiv:2606.24965", + "pdf_url": "https://arxiv.org/pdf/2606.24965", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.23449", + "title": "AOHP: An Open-Source OS-Level Agent Harness for Personalized, Efficient and Secure Interaction", + "url": "https://arxiv.org/abs/2606.23449", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Shanhui Zhao", + "Jiacheng Liu", + "Guohong Liu", + "Jichao Yan", + "Jialei Ye", + "Yuhao Yang", + "Hao Wen", + "Shizuo Tian", + "Yizhen Yuan", + "Yuxuan Chen", + "Yunxin Liu", + "Ju Ren", + "Ya-Qin Zhang", + "Chao Huang", + "Yao Guo", + "Yuanchun Li" + ], + "categories": [ + "cs.AI", + "cs.OS" + ], + "topics": [ + "agent-safety", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.23449", + "source": "arxiv", + "source_id": "arxiv:2606.23449", + "pdf_url": "https://arxiv.org/pdf/2606.23449", + "primary_query": "ai-agent" + }, + { + "id": "2606.23937", + "title": "When Retrieval Metrics Mislead: Measuring Policy Signal in Long-Horizon Tool-Use Agents", + "url": "https://arxiv.org/abs/2606.23937", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Tianyu Ding", + "Juan Pablo De la Cruz Weinstein" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.23937", + "source": "arxiv", + "source_id": "arxiv:2606.23937", + "pdf_url": "https://arxiv.org/pdf/2606.23937", + "primary_query": "tool-use" + }, + { + "id": "2606.23997", + "title": "ChartWalker: Benchmarking the Cross-Chart RAG Task with Hierarchical Knowledge Graphs", + "url": "https://arxiv.org/abs/2606.23997", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Ning Tang", + "Chenghan Xie", + "Hanyang Yuan", + "Yi Li", + "Renhong Huang", + "Qian Kou", + "Xiaofeng Shi", + "Hua Zhou", + "Jiarong Xu" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.23997", + "source": "arxiv", + "source_id": "arxiv:2606.23997", + "pdf_url": "https://arxiv.org/pdf/2606.23997", + "primary_query": "rag-agent" + }, + { + "id": "2606.22376", + "title": "Vibe Calibration: Autonomous Bring-up of a 112-Qubit Superconducting Quantum Processor by a Skill-Orchestrating Language Agent", + "url": "https://arxiv.org/abs/2606.22376", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Huikai Xu", + "Jiaxiu Han", + "Shigang Ou", + "Cheng Ye", + "Zisong Shen", + "Jing Gao", + "Yijia Wang", + "Tianrui Che", + "Yu Song", + "Weiyang Liu", + "Lei Wang", + "Lin-Feng Zhang", + "Pan Zhang", + "Hai-Feng Yu" + ], + "categories": [ + "quant-ph" + ], + "topics": [ + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.22376", + "source": "arxiv", + "source_id": "arxiv:2606.22376", + "pdf_url": "https://arxiv.org/pdf/2606.22376", + "primary_query": "language-agent" + }, + { + "id": "2606.22721", + "title": "Habituation at the Gate: Rising Approval and Declining Scrutiny in Human Review of AI Agent Code", + "url": "https://arxiv.org/abs/2606.22721", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Haoran Yu", + "Lifei Liu", + "Xiaochong Jiang", + "Yuwen Jia", + "Su Wang", + "Pin Qian", + "Yihang Chen" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "coding-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2606.22721", + "source": "arxiv", + "source_id": "arxiv:2606.22721", + "pdf_url": "https://arxiv.org/pdf/2606.22721", + "primary_query": "ai-agent" + }, + { + "id": "2606.22711", + "title": "Beyond Simpson's Paradox: A Cascade of Confounders in AI Agent Pull-Request Co-Authorship", + "url": "https://arxiv.org/abs/2606.22711", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Haoran Yu", + "Xiaochong Jiang", + "Lifei Liu", + "Su Wang", + "Pin Qian", + "Yihang Chen" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2606.22711", + "source": "arxiv", + "source_id": "arxiv:2606.22711", + "pdf_url": "https://arxiv.org/pdf/2606.22711", + "primary_query": "ai-agent" + }, + { + "id": "2606.21843", + "title": "Measuring What Persists: Conditioning Mechanisms and a Geometric Framework for AI Agent Identity", + "url": "https://arxiv.org/abs/2606.21843", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Andrew Tanner" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.21843", + "source": "arxiv", + "source_id": "arxiv:2606.21843", + "pdf_url": "https://arxiv.org/pdf/2606.21843", + "primary_query": "ai-agent" + }, + { + "id": "2606.21562", + "title": "Compressing Observation History into Agent Memory: Distilling Transformers into Recurrent Transformers", + "url": "https://arxiv.org/abs/2606.21562", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Philippe Weinzaepfel", + "Christian Wolf", + "Bülent Mert Sariyildiz", + "Guillaume Bono", + "Gianluca Monaci" + ], + "categories": [ + "cs.CV", + "cs.LG" + ], + "topics": [ + "embodied-agent", + "memory", + "planning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.21562", + "source": "arxiv", + "source_id": "arxiv:2606.21562", + "pdf_url": "https://arxiv.org/pdf/2606.21562", + "primary_query": "agent-memory" + }, + { + "id": "2606.19857", + "title": "Large Language Models Do Not Always Need Readable Language", + "url": "https://arxiv.org/abs/2606.19857", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Jiayi Zhu", + "Haoxuan Peng", + "Junxi Wang", + "Liang Ke", + "Chen Zhang", + "Linfeng Zhang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.19857", + "source": "arxiv", + "source_id": "arxiv:2606.19857", + "pdf_url": "https://arxiv.org/pdf/2606.19857", + "primary_query": "agent-memory" + }, + { + "id": "2606.20910", + "title": "Whose Agent Are You? Multi-Layer Fingerprinting and Attribution of Autonomous Web Agents", + "url": "https://arxiv.org/abs/2606.20910", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Dayeon Kang", + "Hyejun Jeong", + "Jade Sheffey", + "Pubali Datta", + "Amir Houmansadr" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "computer-use", + "embodied-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20910", + "source": "arxiv", + "source_id": "arxiv:2606.20910", + "pdf_url": "https://arxiv.org/pdf/2606.20910", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.20487", + "title": "Beyond Global Replanning: Hierarchical Recovery for Cross-Device Agent Systems", + "url": "https://arxiv.org/abs/2606.20487", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Shu Yao", + "Yuhua Luo", + "Qian Long", + "Jingru Fan", + "Zhuoyuan Yu", + "Yuheng Wang", + "Lin Wu", + "Yufan Dang", + "Huatao Li", + "Chen Qian" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20487", + "source": "arxiv", + "source_id": "arxiv:2606.20487", + "pdf_url": "https://arxiv.org/pdf/2606.20487", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.20746", + "title": "Amplify, Don't Create: Temporal Accumulation for Slow-Burn Prompt Injection", + "url": "https://arxiv.org/abs/2606.20746", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "J Alex Corll" + ], + "categories": [ + "cs.CR", + "cs.LO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation", + "tool-use" + ], + "arxiv_id": "2606.20746", + "source": "arxiv", + "source_id": "arxiv:2606.20746", + "pdf_url": "https://arxiv.org/pdf/2606.20746", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18613", + "title": "Are LLMs Ready to Assist Physicians? PhysAssistBench for Interactive Doctor-Patient-EHR Assistance", + "url": "https://arxiv.org/abs/2606.18613", + "published": "2026-06-17", + "updated": "2026-06-18", + "authors": [ + "Tianming Du", + "Peijie Yu", + "Sihan Shang", + "Danli Shi", + "My Linh Nguyen", + "Shengbo Gao", + "Guangyuan Li", + "Yinghong Yu", + "Yan Jiang", + "Qianlong Zhao", + "Behzad Bozorgtabar", + "Shaoxiong Ji", + "Jiazhen Pan", + "Daniel Rueckert", + "Jiancheng Yang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.18613", + "source": "arxiv", + "source_id": "arxiv:2606.18613", + "pdf_url": "https://arxiv.org/pdf/2606.18613", + "primary_query": "tool-use" + }, + { + "id": "2606.19390", + "title": "Execution-bound advisory automation for agentic AI: a reproducible AIBOM-driven CSAF-VEX framework", + "url": "https://arxiv.org/abs/2606.19390", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Petar Radanliev", + "Omar Santos", + "Carsten Maple", + "Kay Atefi" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.19390", + "source": "arxiv", + "source_id": "arxiv:2606.19390", + "pdf_url": "https://arxiv.org/pdf/2606.19390", + "primary_query": "agentic-ai" + }, + { + "id": "2606.17454", + "title": "Dissecting model behavior through agent trajectories", + "url": "https://arxiv.org/abs/2606.17454", + "published": "2026-06-16", + "updated": "2026-06-17", + "authors": [ + "Gaurav Gupta", + "Vatshank Chaturvedi", + "Jun Huan", + "Anoop Deoras" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.17454", + "source": "arxiv", + "source_id": "arxiv:2606.17454", + "pdf_url": "https://arxiv.org/pdf/2606.17454", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.17929", + "title": "PreAct: Computer-Using Agents that Get Faster on Repeated Tasks", + "url": "https://arxiv.org/abs/2606.17929", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Bojie Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.17929", + "source": "arxiv", + "source_id": "arxiv:2606.17929", + "pdf_url": "https://arxiv.org/pdf/2606.17929", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.16465", + "title": "When Agent Automation Becomes Profitable: Quantifying and Insuring Autonomous AI Risk through Trace-Economic Underwriting", + "url": "https://arxiv.org/abs/2606.16465", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Binyan Xu", + "Xilin Dai", + "Fan Yang", + "Kehuan Zhang" + ], + "categories": [ + "cs.AI", + "cs.CE" + ], + "topics": [ + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.16465", + "source": "arxiv", + "source_id": "arxiv:2606.16465", + "pdf_url": "https://arxiv.org/pdf/2606.16465", + "primary_query": "tool-use" + }, + { + "id": "2606.28365", + "title": "CAMI: Cost-Aware Agent-Guided Multi-Indexing for Semantic Retrieval", + "url": "https://arxiv.org/abs/2606.28365", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Adnan Qidwai", + "Anand Eswaran", + "Sonam Mishra", + "Jaydeep Sen", + "Sachindra Joshi" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.28365", + "source": "arxiv", + "source_id": "arxiv:2606.28365", + "pdf_url": "https://arxiv.org/pdf/2606.28365", + "primary_query": "rag-agent" + }, + { + "id": "2606.15485", + "title": "The Perils of Agency: How Developers Perceive, Prioritize, and Address Risks in Agentic AI Products", + "url": "https://arxiv.org/abs/2606.15485", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Hao-Ping Lee", + "Jessica He", + "David Piorkowski", + "Thomas Serban von Davier", + "Jodi Forlizzi", + "Sauvik Das" + ], + "categories": [ + "cs.CY", + "cs.AI", + "cs.HC", + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.15485", + "source": "arxiv", + "source_id": "arxiv:2606.15485", + "pdf_url": "https://arxiv.org/pdf/2606.15485", + "primary_query": "tool-use" + }, + { + "id": "2606.15335", + "title": "Privacy-Preserving Text Sanitization for Distributed Agents Collaboration via Disentangled Representations", + "url": "https://arxiv.org/abs/2606.15335", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Xuan Liu", + "Hefeng Zhou", + "Sicheng Chen", + "Chao Yang", + "Xingcheng Xu", + "Jingjing Qu", + "Jiong Lou", + "Jie LI", + "Xia Hu" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.15335", + "source": "arxiv", + "source_id": "arxiv:2606.15335", + "pdf_url": "https://arxiv.org/pdf/2606.15335", + "primary_query": "rag-agent" + }, + { + "id": "2606.15007", + "title": "Nemotron 3 Ultra: Open, Efficient Mixture-of-Experts Hybrid Mamba-Transformer Model for Agentic Reasoning", + "url": "https://arxiv.org/abs/2606.15007", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "NVIDIA", + ":", + "Aaron Blakeman", + "Aaron Thomas", + "Aastha Jhunjhunwala", + "Abhibha Gupta", + "Abhinav Khattar", + "Adam Rajfer", + "Adi Renduchintala", + "Adil Asif", + "Aditya Vavre", + "Adriana Flores Miranda", + "Ahmad Bilal", + "Aileen Zaman", + "Ajay Hotchandani", + "Akanksha Shukla", + "Akhiad Bercovich", + "Aleksander Ficek", + "Alex Gronskiy", + "Alex Kondratenko", + "Alex Steiner", + "Alex Ye", + "Alexander Bukharin", + "Alexandre Milesi", + "Ali Taghibakhshi", + "Alice Gatti", + "Alisa Liu", + "Alok Kumar", + "Amar Phanishayee", + "Ameya Sunil Mahabaleshwarkar", + "Amir Klein", + "Amit Zuker", + "Amnon Geifman", + "Anahita Bhiwandiwalla", + "Ananth Subramaniam", + "Andrea Santilli", + "Andrew Fulks", + "Andrew McHarg", + "Andrew Tao", + "Andrii Skliar", + "Anjulie Agrusa", + "Ankur Srivastava", + "Ankur Verma", + "Anna Shors", + "Anna Warno", + "Antoni-Joan Solergibert I Llaquet", + "Arham Mehta", + "Arkadiusz Nowaczynski", + "Arti Jain", + "Ashwath Aithal", + "Ashwin Poojary", + "Asif Ahamed", + "Asit Mishra", + "Asma Kuriparambil Thekkumpate", + "Atefeh Sohrabizadeh", + "Avinash Kaur", + "Avinash Vem", + "Ayush Dattagupta", + "Barath Subramaniam Anandan", + "Bardiya Sadeghi", + "Ben Lanir", + "Benedikt Schifferer", + "Besmira Nushi", + "Bilal Kartal", + "Bill Thiede", + "Bita Darvish Rouhani", + "Bo Deng", + "Bob Schatz", + "Boris Ginsburg", + "Boxin Wang", + "Brad Nemire", + "Brandon Norick", + "Brian Dang", + "Brian Westphal", + "Brian Yu", + "Brucek Khailany", + "Bryan Catanzaro", + "Carlo del Mundo", + "Caryln Aarish", + "Chankyu Lee", + "Chantal Hwang", + "Charbel Sakr", + "Charles Wang", + "Charlie Truong", + "Chen Cui", + "Cheng Cheng", + "Cheng-Ping Hsieh", + "Chenghao Zhang", + "Chenhui Deng", + "Chintan Patel", + "Chris Alexiuk", + "Christian Cosgrove", + "Christian Munley", + "Christine Harvey", + "Christopher Parisien", + "Chunyang Shen", + "Coco Li", + "Collin Neale", + "Cynthia Gao", + "Cyril Meurillon", + "Dan Gil", + "Dan Su", + "Dan Zhao", + "Dane Corneil", + "Daniel Afrimi", + "Daniel Egert", + "Daniel Korzekwa", + "Daniel Lo", + "Daniel Machlab", + "Daniel Serebrenik", + "Daniil Sorokin", + "Daria Gitman", + "Daria Levy", + "Darko Stosic", + "David Mosallanezhad", + "David Yu", + "Davit Karamyan", + "Deena Donia", + "Deep Debroy", + "Deepak Narayanan", + "Devin O'Kelly", + "Dheeraj Peri", + "Dhruv Nathawani", + "Di", + "Wu", + "Dima Rekesh", + "Divyanshu Kakwani", + "Donald Plummer", + "Dong Anh", + "Dongfeng Yu", + "Dongfu Jiang", + "Donnie Kim", + "Dorrin Poorkay", + "Duncan Riach", + "Dusan Stosic", + "Dustin VanStee", + "Eavan Meng", + "Edgar Minasyan", + "Edward Lin", + "Eileen Margaret Peters Long", + "Elad Sarafin", + "Elad Segal", + "Elena Lantz", + "Ellie Evans", + "Elliott Ning", + "Eric Chung", + "Eric Harper", + "Eric Pham-Hung", + "Eric Tramel", + "Eric Yang", + "Erick Galinkin", + "Erik Pounds", + "Erika Goncalves Goncalves", + "Evan Briones", + "Evan Wu", + "Evelina Bakhturina", + "Evgeny Tsykunov", + "Ewa Dobrowolska", + "Faisal Ladhak", + "Farzan Memarian", + "Fay Wang", + "Fei Jia", + "Felipe Soares", + "Felipe Vieira Frujeri", + "Feng Chen", + "Fengguang Lin", + "Ferenc Galko", + "Frank Sun", + "Frankie Siino", + "Frida Hou", + "Gal Hubara Agam", + "Gal Kaplun", + "Gantavya Bhatt", + "Gargi Prasad", + "Garvit Kulshreshtha", + "George Armstrong", + "Gerald Shen", + "Giulio Borghesi", + "Gordana Neskovic", + "Gorkem Batmaz", + "Grace Lam", + "Greg Mason", + "Greg Pauloski", + "Grigor Nalbandyan", + "Grzegorz Chlebus", + "Grzegorz Karch", + "Guan-Ting Liu", + "Guoming Zhang", + "Guyue Huang", + "Haggai Maron", + "Haifeng Qian", + "Haim Elisha", + "Haoxing Ren", + "Haran Kumar Shiv Kumar", + "Haribhau Hud", + "Harris Nover", + "Harrison Saturley Hall", + "Hayate Iso", + "Helen Ngo", + "Herbert Hum", + "Herman Sahota", + "Hexin Wang", + "Himanshu Soni", + "Hovhannes Tamoyan", + "Hua Li", + "Huanhuan Chen", + "Hui Li", + "Hui Wang", + "Huy Nguyen", + "Ian Chiles", + "Ido Galil", + "Ido Shahaf", + "Igor Gitman", + "Igor Shovkun", + "Ilya Loshchilov", + "Ingo Guehring", + "Itamar Schen", + "Itay Levy", + "Itay Neeman", + "Ivan Moshkov", + "Izik Golan", + "Izzy Putterman", + "Jaemin Choi", + "Jakub Slowikowski", + "Jan Kautz", + "Jane Polak Scowcroft", + "Jared Casper", + "Jatin Mitra", + "Jeffrey Glick", + "Jenny Chen", + "Jesse Oliver", + "Jiacheng Xu", + "Jiafan Zhu", + "Jialin Song", + "Jian Zhang", + "Jiantao Jiao", + "Jiaqi Zeng", + "Jie Lou", + "Jim King", + "Jimmy Zhang", + "Jingquan Wang", + "Jinhang Choi", + "Jinju Chu", + "Joey Conway", + "Joey Guman", + "Johan Jatko", + "Johannes Rausch", + "John Kamalu", + "John Roberts", + "Johnny Greco", + "Johnny Mensel", + "Jonah Alben", + "Jonas Yang", + "Jonathan Cohen", + "Jonathan Raiman", + "Joseph Jennings", + "Joshua Mabry", + "Joshua Pierce", + "Joyjit Daw", + "Julien Veron Vialard", + "Junkeun Yi", + "Jupinder Parmar", + "Kajal Jain", + "Kan Zhu", + "Kari Briski", + "Katherine Cheung", + "Katherine Luna", + "Keith Willowhawk", + "Keith Wyss", + "Keshav Santhanam", + "Kevin Shih", + "Kezhi Kong", + "Khanh Nguyen", + "Khushi Bhardwaj", + "Kirthi Shankar Sivamani", + "Konstantinos Krommydas", + "Krishna C. Puvvada", + "Krzysztof Pawelec", + "Kumar Anik", + "Kyle Keprios", + "Kylie Day", + "Lawrence McAfee", + "Leo Du", + "Leon Derczynski", + "Li Ding", + "Linda Liu", + "Lingjie Wu", + "Lior Kadoch", + "Lizzie Wei", + "Luis Vega", + "Luke Robison", + "Lun Su", + "Maarten Van Segbroeck", + "Maciej Jakub Mikulski", + "Maer Rodrigues de Melo", + "Magda Sypula", + "Mahan Fathi", + "Makesh Narsimhan Sreedhar", + "Makesh Tarun Chandran", + "Manoj Kilaru", + "Maor Ashkenazi", + "Marc Cuevas", + "Marc Romeijn", + "Marcin Chochowski", + "Mark Cai", + "Mark Mozolewski", + "Markus Kliegl", + "Marta Stepniewska-Dziubinska", + "Martyna Patelka", + "Mattei Machczynski", + "Matvei Novikov", + "Mauricio Ferrato", + "Maximilian Golub", + "Mehrzad Samadi", + "Melissa Corpuz", + "Mengru Wang", + "Mengxi Wu", + "Meredith Price", + "Meriem Boubdir", + "Micah Schaffer", + "Michael Andersch", + "Michael Boone", + "Michael Gschwind", + "Michael Lightstone", + "Michael Loh", + "Michal Bien", + "Michal Zawalski", + "Michelle Gill", + "Miguel Martinez", + "Mikail Khona", + "Mike Chrzanowski", + "Mike Houston", + "Mingyuan Ma", + "Minseok Lee", + "Mohamed Fawzy", + "Mohammad Dabbah", + "Mohammad Shoeybi", + "Mostofa Patwary", + "Nabin Mulepati", + "Najeeb Nabwani", + "Namit Dhameja", + "Narimane Hennouni", + "Natalie Hereth", + "Nathaniel Pinckney", + "Nave Algarici", + "Nave Assaf", + "Netanel Haber", + "Nicholas Knight", + "Nick Reamaroon", + "Nickson Quak", + "Nidhi Bhatia", + "Nikhil Desai", + "Nikolai Ludwig", + "Nima Tajbakhsh", + "Ning Xu", + "Nir Ailon", + "Nirmal Juluru", + "Nitin Nitin", + "Ofri Masad", + "Oleg Rybakov", + "Oleksii Hrinchuk", + "Oleksii Kuchaiev", + "Olivia Viessmann", + "Olivier Delalleau", + "Oluwatobi Olabiyi", + "Omer Ullman Argov", + "Omri Puny", + "Oren Tropp", + "Pablo Ribalta", + "Pallab Bhattacharya", + "Panos Lampropoulos", + "Parth Mannan", + "Pasha Shamis", + "Patrick Legresley", + "Paul Gibbons", + "Pavlo Molchanov", + "Pawel Morkisz", + "Peter Dykas", + "Peter Jin", + "Pierre-Yves Aquilanti", + "Pinky Xu", + "Piotr Januszewski", + "Piotr Laskiewicz", + "Pooya Jannaty", + "Prakash Gurumurthy", + "Pranav Prashant Thombre", + "Prasoon Varshney", + "Pritam Gundecha", + "Przemek Tredak", + "Puhui Meng", + "Qiyu Wan", + "Rabeeh Karimi Mahabadi", + "Rachel Oberman", + "Rachit Garg", + "Radha Sri-Tharan", + "Rahul Kandu", + "Rakshit Sanadhya", + "Ran El-Yaniv", + "Ran Zilberstein", + "Rasoul Shafipour", + "Ray Macalisang", + "Rayen Tian", + "Reka Kovacs", + "Renjie Pi", + "Rick Izzo", + "Rima Shahbazyan", + "Rishabh Garg", + "Rishi Puri", + "Rita Fernandes Neves", + "Ritchie Zhao", + "Ritika Borkar", + "Ritu Gala", + "Riyad Islam", + "Robert Clark", + "Robert Hesse", + "Robert Kirby", + "Roger Waleffe", + "Rohit Watve", + "Roi Koren", + "Ron Banner", + "Ruoxi Zhang", + "Russell J. Hewett", + "Ryan Prenger", + "Ryan Stewart", + "Ryota Egashira", + "Sadegh Mahdavi", + "Saee Paliwal", + "Sagar Singh", + "Sahil Modi", + "Salika Dave", + "Samantha Shinagawa", + "Samuel Kriman", + "Sandip Bhaskar", + "Sangkug Lym", + "Sanjay Kariyappa", + "Sanjeev Satheesh", + "Saran Vikas Murari", + "Satish Pasumarthi", + "Saurabh Mishra", + "Saurav Muralidharan", + "Scott Hara", + "Sean Narentharen", + "Selvaraj Anandaraj", + "Seonjin Na", + "Seonmeyong Bak", + "Seonmyeong Bak", + "Sepehr Sameni", + "Seph Mard", + "Serge Panev", + "Seth Henneman", + "Seth Poulos", + "Shahar Mor", + "Shantanu Acharya", + "Shaona Ghosh", + "Sharath Turuvekere Sreenivas", + "Sharon Mendelson", + "Shaun Kotek", + "Shawn Wang", + "Shay Aharon", + "Shaya Gharghabi", + "Sheng-Chieh Lin", + "Shi Chen", + "Shiqing Fan", + "Shirish Baskaran", + "Shreya Gopa", + "Shrimai Prabhumoye", + "Shubham Pachori", + "Shubham Toshniwal", + "Shuoyang Ding", + "Shwetha Krishnamurthy", + "Siddharth Singh", + "Simeng Sun", + "Sirshak Das", + "Sivakumar Arayandi Thottakara", + "Smita Ithape", + "Somshubra Majumdar", + "Soumye Singhal", + "Sri Harsha Singudasu", + "Sridhar Bhuvanapalli", + "Srimukh Veccham", + "Stas Sergienko", + "Stefania Alborghetti", + "Stephen Ge", + "Su Rong", + "Sugam Dipak Devare", + "Sukrit Rao", + "Sumeet Kumar Barua", + "Sungsoo Ha", + "Sunny Gai", + "Suriya Gunasekar", + "Suseella Panguluri", + "Suyog Gupta", + "Sviataslau Hinzburh", + "Sweta Priyadarshi", + "Syeda Nahida Akter", + "Talor Abramovich", + "Tan Bui", + "Tanay Varshney", + "Tatevik Ter-Hovhannisyan", + "Teodor-Dumitru Ene", + "Terry Kong", + "Thanh Do", + "Tianhe Zhang", + "Tiffany Moore", + "Tijmen Blankevoort", + "Tim Moon", + "Tiyasa Mitra", + "Tom Balough", + "Tomasz Grzegorzek", + "Tomasz Hliwiak", + "Tomer Asida", + "Tomer Bar Natan", + "Tomer Keren", + "Tomer Ronen", + "Tony Salim", + "Tony Wang", + "Traian Rebedea", + "Tugrul Konuk", + "Twinkle Vashishth", + "Udi Karpas", + "Ushnish De", + "Vahid Noorozi", + "Venkat Srinivasan", + "Venmugil Elango", + "Vibhor Agrawal", + "Victor Cui", + "Vijay Korthikanti", + "Vikas Mehta", + "Vinay Rao", + "Virginia Wu", + "Vitaly Kurin", + "Vitaly Lavrukhin", + "Vladimir Anisimov", + "Vu Pham", + "Wanli Jiang", + "Wasi Uddin Ahmad", + "Wataru Ishihara", + "Wei Du", + "Wei Ping", + "Weiheng Chai", + "Wenliang Dai", + "Wesley Helmholz", + "Will Jennings", + "Will Zhu", + "Wojciech Prazuch", + "Xiaowei Ren", + "Xiwen Yu", + "Yan Breek", + "Yang Chen", + "Yang Yu", + "Yangyi Chen", + "Yaniv Galron", + "Yashaswi Karnati", + "Yejin Choi", + "Yev Meyer", + "Yi-Fu Wu", + "Yian Zhang", + "Ying Lin", + "Yonatan Geifman", + "Yonggan Fu", + "Youngeun Kwon", + "Yu Yao", + "Yugi Guvvla", + "Yuki Huang", + "Yunsheng Liu", + "Zach Moshe", + "Zachary Newell", + "Zhilin Wang", + "Zhiyu Li", + "Zhongbo Zhu", + "Zhuolin Yang", + "Zihan Liu", + "Zijie Yan", + "Zsolt-Alon Wertheimer" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.15007", + "source": "arxiv", + "source_id": "arxiv:2606.15007", + "pdf_url": "https://arxiv.org/pdf/2606.15007", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.14266", + "title": "Large Language Model Based Agent for Automated Discovery in Computational Physics", + "url": "https://arxiv.org/abs/2606.14266", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Hang Lin", + "Chongwen Liu", + "Gang Yan" + ], + "categories": [ + "physics.comp-ph" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.14266", + "source": "arxiv", + "source_id": "arxiv:2606.14266", + "pdf_url": "https://arxiv.org/pdf/2606.14266", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.12830", + "title": "Perceive, Interact, Reason: Building Tool-Augmented Visual Agents for Spatial Reasoning", + "url": "https://arxiv.org/abs/2606.12830", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Changye Li", + "Meng Lu", + "Yi Wu", + "Ligeng Zhu" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.12830", + "source": "arxiv", + "source_id": "arxiv:2606.12830", + "pdf_url": "https://arxiv.org/pdf/2606.12830", + "primary_query": "tool-use" + }, + { + "id": "2606.13380", + "title": "An LLM System for Autonomous Variational Quantum Circuit Design", + "url": "https://arxiv.org/abs/2606.13380", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Kenya Sakka", + "Wataru Mizukami", + "Kosuke Mitarai" + ], + "categories": [ + "quant-ph", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.13380", + "source": "arxiv", + "source_id": "arxiv:2606.13380", + "pdf_url": "https://arxiv.org/pdf/2606.13380", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.12666", + "title": "CAPED: Context-Aware Privacy Exposure Defense for Mobile GUI Agents", + "url": "https://arxiv.org/abs/2606.12666", + "published": "2026-06-10", + "updated": "2026-06-16", + "authors": [ + "Siyu Shen", + "Fenghao Xu", + "Wenrui Diao", + "Kehuan Zhang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.12666", + "source": "arxiv", + "source_id": "arxiv:2606.12666", + "pdf_url": "https://arxiv.org/pdf/2606.12666", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.10522", + "title": "GUI-AC: Enhancing Continual Learning in GUI Agents", + "url": "https://arxiv.org/abs/2606.10522", + "published": "2026-06-09", + "updated": "2026-07-06", + "authors": [ + "Can Lin", + "Tao Feng", + "Hangjie Yuan", + "Dan Zhang", + "Yifan Zhu", + "Zhonghong Ou" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "computer-use", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.10522", + "source": "arxiv", + "source_id": "arxiv:2606.10522", + "pdf_url": "https://arxiv.org/pdf/2606.10522", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.10062", + "title": "Deployment-Time Memorization in Foundation-Model Agents", + "url": "https://arxiv.org/abs/2606.10062", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Lei", + "Chen", + "Guilin Zhang", + "Kai Zhao", + "Dalmo Cirne", + "Andy Olsen", + "Xu Chu", + "Zeke Miller", + "Alet Blanken", + "Amine Anoun", + "Jerry Ting" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.10062", + "source": "arxiv", + "source_id": "arxiv:2606.10062", + "pdf_url": "https://arxiv.org/pdf/2606.10062", + "primary_query": "agent-memory" + }, + { + "id": "2606.06708", + "title": "Signal-Driven Observation for Long-Horizon Web Agents", + "url": "https://arxiv.org/abs/2606.06708", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Shubham Gaur", + "Ian Lane" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.06708", + "source": "arxiv", + "source_id": "arxiv:2606.06708", + "pdf_url": "https://arxiv.org/pdf/2606.06708", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.04391", + "title": "Online Skill Learning for Web Agents via State-Grounded Dynamic Retrieval", + "url": "https://arxiv.org/abs/2606.04391", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Jiaxi Li", + "Ke Deng", + "Yun Wang", + "Jingyuan Huang", + "Yucheng Shi", + "Qiaoyu Tan", + "Jin Lu", + "Ninghao Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.04391", + "source": "arxiv", + "source_id": "arxiv:2606.04391", + "pdf_url": "https://arxiv.org/pdf/2606.04391", + "primary_query": "language-agent" + }, + { + "id": "2606.05112", + "title": "Evaluating Large Language Models in Dynamic Clinical Decision-Making with Standardized Patient Cases", + "url": "https://arxiv.org/abs/2606.05112", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Cheng Liang", + "Pengcheng Qiu", + "Ya Zhang", + "Yanfeng Wang", + "Chaoyi Wu", + "Weidi Xie" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "world-model" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.05112", + "source": "arxiv", + "source_id": "arxiv:2606.05112", + "pdf_url": "https://arxiv.org/pdf/2606.05112", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.04435", + "title": "Cascading Hallucination in Agentic RAG: The CHARM Framework for Detection and Mitigation", + "url": "https://arxiv.org/abs/2606.04435", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Saroj Mishra" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CR", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.04435", + "source": "arxiv", + "source_id": "arxiv:2606.04435", + "pdf_url": "https://arxiv.org/pdf/2606.04435", + "primary_query": "rag-agent" + }, + { + "id": "2606.28347", + "title": "Agentic Safety is an Epistemic Property, Not a Behavioral One", + "url": "https://arxiv.org/abs/2606.28347", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Charles L. Wang", + "Keir Dorchen", + "Peter Jin" + ], + "categories": [ + "cs.CY", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-safety", + "embodied-agent", + "rag" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.28347", + "source": "arxiv", + "source_id": "arxiv:2606.28347", + "pdf_url": "https://arxiv.org/pdf/2606.28347", + "primary_query": "agent-safety" + }, + { + "id": "2606.28344", + "title": "PIXELRAG: Web Screenshots Beat Text for Retrieval-Augmented Generation", + "url": "https://arxiv.org/abs/2606.28344", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Yichuan Wang", + "Zhifei Li", + "Zirui Wang", + "Paul Teiletche", + "Lesheng Jin", + "Matei Zaharia", + "Joseph E. Gonzalez", + "Sewon Min" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.CL", + "cs.CV", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation", + "rag-agent" + ], + "arxiv_id": "2606.28344", + "source": "arxiv", + "source_id": "arxiv:2606.28344", + "pdf_url": "https://arxiv.org/pdf/2606.28344", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.00183", + "title": "Agentic Transformers Provably Learn to Search via Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.00183", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Tong Yang", + "Yu Huang", + "Yingbin Liang", + "Yuejie Chi" + ], + "categories": [ + "cs.LG", + "cs.AI", + "math.OC", + "stat.ML" + ], + "topics": [ + "memory", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.00183", + "source": "arxiv", + "source_id": "arxiv:2606.00183", + "pdf_url": "https://arxiv.org/pdf/2606.00183", + "primary_query": "language-agent" + }, + { + "id": "2605.30916", + "title": "Welfare, Improvability, and Variance: A Principal-Agent Approach to Optimal Benchmark Item Aggregation", + "url": "https://arxiv.org/abs/2605.30916", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Andreas Haupt", + "Justin Hartenstein", + "Anka Reuel", + "Mykel Kochenderfer", + "Sanmi Koyejo" + ], + "categories": [ + "cs.LG", + "cs.GT", + "econ.TH" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.30916", + "source": "arxiv", + "source_id": "arxiv:2605.30916", + "pdf_url": "https://arxiv.org/pdf/2605.30916", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.29786", + "title": "Croissant Tasks: A Metadata Format for Reproducible Machine Learning Evaluations", + "url": "https://arxiv.org/abs/2605.29786", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Omar Benjelloun", + "Leonardo Martins Bianco", + "Isabelle Guyon", + "Thanh Gia Hieu Khuong", + "Jonathan Lebensold", + "Sebastian Lobentanzer", + "Luis Oala", + "Benedictus Kent Rachmat", + "Ihsan Ullah", + "Peyman Vahidi", + "Joaquin Vanschoren" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.29786", + "source": "arxiv", + "source_id": "arxiv:2605.29786", + "pdf_url": "https://arxiv.org/pdf/2605.29786", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.27995", + "title": "AsyncTool: Evaluating the Asynchronous Function Calling Capability under Multi-Task Scenarios", + "url": "https://arxiv.org/abs/2605.27995", + "published": "2026-05-27", + "updated": "2026-05-28", + "authors": [ + "Kou Shi", + "Ziao Zhang", + "Shiting Huang", + "Avery Nie", + "Zhen Fang", + "Qiuchen Wang", + "Lin Chen", + "Huaian Chen", + "Zehui Chen", + "Feng Zhao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.27995", + "source": "arxiv", + "source_id": "arxiv:2605.27995", + "pdf_url": "https://arxiv.org/pdf/2605.27995", + "primary_query": "function-calling" + }, + { + "id": "2605.23459", + "title": "AI Assurance: A Comprehensive Testing Strategy for Enterprise AI Systems", + "url": "https://arxiv.org/abs/2605.23459", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Chitra Badagi", + "Divye Singh", + "Animesh Sen", + "Adinath Shirsath" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.23459", + "source": "arxiv", + "source_id": "arxiv:2605.23459", + "pdf_url": "https://arxiv.org/pdf/2605.23459", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.20896", + "title": "GenAI-Driven Threat Detection with Microsoft Security Copilot", + "url": "https://arxiv.org/abs/2605.20896", + "published": "2026-05-20", + "updated": "2026-05-22", + "authors": [ + "Scott Freitas", + "Amir Gharib" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.20896", + "source": "arxiv", + "source_id": "arxiv:2605.20896", + "pdf_url": "https://arxiv.org/pdf/2605.20896", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.20477", + "title": "Training Language Agents to Learn from Experience", + "url": "https://arxiv.org/abs/2605.20477", + "published": "2026-05-19", + "updated": "2026-05-19", + "authors": [ + "Yuval Shalev", + "Zifeng Ding", + "Mateja Jamnik" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.20477", + "source": "arxiv", + "source_id": "arxiv:2605.20477", + "pdf_url": "https://arxiv.org/pdf/2605.20477", + "primary_query": "language-agent" + }, + { + "id": "2605.15077", + "title": "Concurrency without Model Changes: Future-based Asynchronous Function Calling for LLMs", + "url": "https://arxiv.org/abs/2605.15077", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Guangyu Feng", + "Huanzhi Mao", + "Prabal Dutta", + "Joseph E. Gonzalez" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.15077", + "source": "arxiv", + "source_id": "arxiv:2605.15077", + "pdf_url": "https://arxiv.org/pdf/2605.15077", + "primary_query": "function-calling" + }, + { + "id": "2605.12460", + "title": "Multi-Stream LLMs: Unblocking Language Models with Parallel Streams of Thoughts, Inputs and Outputs", + "url": "https://arxiv.org/abs/2605.12460", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Guinan Su", + "Yanwu Yang", + "Xueyan Li", + "Jonas Geiping" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.12460", + "source": "arxiv", + "source_id": "arxiv:2605.12460", + "pdf_url": "https://arxiv.org/pdf/2605.12460", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.11442", + "title": "Can a Single Message Paralyze the AI Infrastructure? The Rise of AbO-DDoS Attacks through Targeted Mobius Injection", + "url": "https://arxiv.org/abs/2605.11442", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Zi Liang", + "Ronghua Li", + "Yanyun Wang", + "Qingqing Ye", + "Haibo Hu" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.11442", + "source": "arxiv", + "source_id": "arxiv:2605.11442", + "pdf_url": "https://arxiv.org/pdf/2605.11442", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.11235", + "title": "Internalizing Curriculum Judgment for LLM Reinforcement Fine-Tuning", + "url": "https://arxiv.org/abs/2605.11235", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Han Zheng", + "Yining Ma", + "Karthick Gunasekaran", + "Bharathan Balaji", + "Zheng Du", + "Shiv Vitaladevuni", + "Cathy Wu" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.11235", + "source": "arxiv", + "source_id": "arxiv:2605.11235", + "pdf_url": "https://arxiv.org/pdf/2605.11235", + "primary_query": "function-calling" + }, + { + "id": "2605.09990", + "title": "Merlin: Deterministic Byte-Exact Deduplication for Lossless Context Optimization in Large Language Model Inference", + "url": "https://arxiv.org/abs/2605.09990", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Sietse Schelpe" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.09990", + "source": "arxiv", + "source_id": "arxiv:2605.09990", + "pdf_url": "https://arxiv.org/pdf/2605.09990", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.06320", + "title": "Improving the Efficiency of Language Agent Teams with Adaptive Task Graphs", + "url": "https://arxiv.org/abs/2605.06320", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Elizabeth Mieczkowski", + "Alexander Ku", + "Tiwalayo Eisape", + "Dilip Arumugam", + "John Matters", + "Katherine M. Collins", + "Ilia Sucholutsky", + "Thomas L. Griffiths" + ], + "categories": [ + "cs.MA", + "cs.AI", + "cs.CL" + ], + "topics": [ + "planning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.06320", + "source": "arxiv", + "source_id": "arxiv:2605.06320", + "pdf_url": "https://arxiv.org/pdf/2605.06320", + "primary_query": "language-agent" + }, + { + "id": "2605.06161", + "title": "Beyond Accuracy: Policy Invariance as a Reliability Test for LLM Safety Judges", + "url": "https://arxiv.org/abs/2605.06161", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Shihao Weng", + "Yang Feng", + "Xiaofei Xie" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.06161", + "source": "arxiv", + "source_id": "arxiv:2605.06161", + "pdf_url": "https://arxiv.org/pdf/2605.06161", + "primary_query": "agent-safety" + }, + { + "id": "2604.27264", + "title": "Self-Evolving Software Agents", + "url": "https://arxiv.org/abs/2604.27264", + "published": "2026-04-29", + "updated": "2026-04-29", + "authors": [ + "Marco Robol", + "Paolo Giorgini" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27264", + "source": "arxiv", + "source_id": "arxiv:2604.27264", + "pdf_url": "https://arxiv.org/pdf/2604.27264", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.10516", + "title": "Structure-Grounded Knowledge Retrieval via Code Dependencies for Multi-Step Data Reasoning", + "url": "https://arxiv.org/abs/2604.10516", + "published": "2026-04-12", + "updated": "2026-04-26", + "authors": [ + "Xinyi Huang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2604.10516", + "source": "arxiv", + "source_id": "arxiv:2604.10516", + "pdf_url": "https://arxiv.org/pdf/2604.10516", + "primary_query": "function-calling" + }, + { + "id": "2604.07960", + "title": "TOOLCAD: Exploring Tool-Using Large Language Models in Text-to-CAD Generation with Reinforcement Learning", + "url": "https://arxiv.org/abs/2604.07960", + "published": "2026-04-09", + "updated": "2026-04-20", + "authors": [ + "Yifei Gong", + "Xing Wu", + "Wenda Liu", + "Kang Tu" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.CL" + ], + "topics": [ + "planning", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.07960", + "source": "arxiv", + "source_id": "arxiv:2604.07960", + "pdf_url": "https://arxiv.org/pdf/2604.07960", + "primary_query": "language-agent" + }, + { + "id": "2603.25723", + "title": "Natural-Language Agent Harnesses", + "url": "https://arxiv.org/abs/2603.25723", + "published": "2026-03-26", + "updated": "2026-05-18", + "authors": [ + "Linyue Pan", + "Lexiao Zou", + "Shuo Guo", + "Jingchen Ni", + "Hai-Tao Zheng" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.25723", + "source": "arxiv", + "source_id": "arxiv:2603.25723", + "pdf_url": "https://arxiv.org/pdf/2603.25723", + "primary_query": "language-agent" + }, + { + "id": "2603.17170", + "title": "PAuth - Precise Task-Scoped Authorization For Agents", + "url": "https://arxiv.org/abs/2603.17170", + "published": "2026-03-17", + "updated": "2026-03-17", + "authors": [ + "Reshabh K Sharma", + "Linxi Jiang", + "Zhiqiang Lin", + "Shuo Chen" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.PL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.17170", + "source": "arxiv", + "source_id": "arxiv:2603.17170", + "pdf_url": "https://arxiv.org/pdf/2603.17170", + "primary_query": "agent-safety" + }, + { + "id": "2602.22523", + "title": "Cognitive Models and AI Algorithms Provide Templates for Designing Language Agents", + "url": "https://arxiv.org/abs/2602.22523", + "published": "2026-02-26", + "updated": "2026-02-26", + "authors": [ + "Ryan Liu", + "Dilip Arumugam", + "Cedegao E. Zhang", + "Sean Escola", + "Xaq Pitkow", + "Thomas L. Griffiths" + ], + "categories": [ + "cs.AI", + "cs.CL", + "q-bio.NC" + ], + "topics": [ + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.22523", + "source": "arxiv", + "source_id": "arxiv:2602.22523", + "pdf_url": "https://arxiv.org/pdf/2602.22523", + "primary_query": "language-agent" + }, + { + "id": "2602.14364", + "title": "A Trajectory-Based Safety Audit of Clawdbot (OpenClaw)", + "url": "https://arxiv.org/abs/2602.14364", + "published": "2026-02-16", + "updated": "2026-02-16", + "authors": [ + "Tianyu Chen", + "Dongrui Liu", + "Xia Hu", + "Jingyi Yu", + "Wenjie Wang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.14364", + "source": "arxiv", + "source_id": "arxiv:2602.14364", + "pdf_url": "https://arxiv.org/pdf/2602.14364", + "primary_query": "agent-safety" + }, + { + "id": "2602.10007", + "title": "A Collaborative Safety Shield for Safe and Efficient CAV Lane Changes in Congested On-Ramp Merging", + "url": "https://arxiv.org/abs/2602.10007", + "published": "2026-02-10", + "updated": "2026-02-10", + "authors": [ + "Bharathkumar Hegde", + "Melanie Bouroche" + ], + "categories": [ + "cs.RO", + "cs.AI", + "cs.MA", + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.10007", + "source": "arxiv", + "source_id": "arxiv:2602.10007", + "pdf_url": "https://arxiv.org/pdf/2602.10007", + "primary_query": "agent-safety" + }, + { + "id": "2604.09554", + "title": "LABBench2: An Improved Benchmark for AI Systems Performing Biology Research", + "url": "https://arxiv.org/abs/2604.09554", + "published": "2026-02-04", + "updated": "2026-05-05", + "authors": [ + "Jon M Laurent", + "Albert Bou", + "Michael Pieler", + "Conor Igoe", + "Alex Andonian", + "Siddharth Narayanan", + "James Braza", + "Alexandros Sanchez Vassopoulos", + "Jacob L Steenwyk", + "Blake Lash", + "Andrew D White", + "Samuel G Rodriques" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.09554", + "source": "arxiv", + "source_id": "arxiv:2604.09554", + "pdf_url": "https://arxiv.org/pdf/2604.09554", + "primary_query": "language-agent" + }, + { + "id": "2603.00030", + "title": "SimpleTool: Parallel Decoding for Real-Time LLM Function Calling", + "url": "https://arxiv.org/abs/2603.00030", + "published": "2026-02-04", + "updated": "2026-02-04", + "authors": [ + "Xiaoxin Shi", + "Jiaxin Wan", + "Linkang Dong", + "Wei Jiang", + "Yue Liu", + "Zengfeng Huang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "embodied-agent", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.00030", + "source": "arxiv", + "source_id": "arxiv:2603.00030", + "pdf_url": "https://arxiv.org/pdf/2603.00030", + "primary_query": "function-calling" + }, + { + "id": "2601.18282", + "title": "Think-Augmented Function Calling: Improving LLM Parameter Accuracy Through Embedded Reasoning", + "url": "https://arxiv.org/abs/2601.18282", + "published": "2026-01-26", + "updated": "2026-02-06", + "authors": [ + "Lei Wei", + "Xiao Peng", + "Jinpeng Ou", + "Bin Wang" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.18282", + "source": "arxiv", + "source_id": "arxiv:2601.18282", + "pdf_url": "https://arxiv.org/pdf/2601.18282", + "primary_query": "function-calling" + }, + { + "id": "2603.21013", + "title": "A Framework for Low-Latency, LLM-driven Multimodal Interaction on the Pepper Robot", + "url": "https://arxiv.org/abs/2603.21013", + "published": "2026-01-09", + "updated": "2026-01-09", + "authors": [ + "Erich Studerus", + "Vivienne Jia Zhong", + "Stephan Vonschallen" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.RO" + ], + "topics": [ + "computer-use", + "embodied-agent", + "planning", + "rag", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.21013", + "source": "arxiv", + "source_id": "arxiv:2603.21013", + "pdf_url": "https://arxiv.org/pdf/2603.21013", + "primary_query": "function-calling" + }, + { + "id": "2512.11277", + "title": "When Actions Teach You to Think: Reasoning-Action Synergy via Reinforcement Learning in Conversational Agents", + "url": "https://arxiv.org/abs/2512.11277", + "published": "2025-12-12", + "updated": "2025-12-12", + "authors": [ + "Mrinal Rawat", + "Arkajyoti Chakraborty", + "Neha Gupta", + "Roberto Pieraccini" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.11277", + "source": "arxiv", + "source_id": "arxiv:2512.11277", + "pdf_url": "https://arxiv.org/pdf/2512.11277", + "primary_query": "function-calling" + }, + { + "id": "2512.01270", + "title": "Egent: An Autonomous Agent for Equivalent Width Measurement", + "url": "https://arxiv.org/abs/2512.01270", + "published": "2025-12-01", + "updated": "2026-05-06", + "authors": [ + "Yuan-Sen Ting", + "Serat Mahmud Saad", + "Fan Liu", + "Yuting Shen" + ], + "categories": [ + "astro-ph.IM", + "astro-ph.GA", + "astro-ph.SR" + ], + "topics": [ + "rag", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.01270", + "source": "arxiv", + "source_id": "arxiv:2512.01270", + "pdf_url": "https://arxiv.org/pdf/2512.01270", + "primary_query": "function-calling" + }, + { + "id": "2509.00482", + "title": "Talk Less, Call Right: Enhancing Role-Play LLM Agents with Automatic Prompt Optimization and Role Prompting", + "url": "https://arxiv.org/abs/2509.00482", + "published": "2025-08-30", + "updated": "2025-10-12", + "authors": [ + "Saksorn Ruangtanusak", + "Pittawat Taveekitworachai", + "Kunat Pipatanakul" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.HC" + ], + "topics": [ + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.00482", + "source": "arxiv", + "source_id": "arxiv:2509.00482", + "pdf_url": "https://arxiv.org/pdf/2509.00482", + "primary_query": "function-calling" + }, + { + "id": "2508.20931", + "title": "How Can Input Reformulation Improve Tool Usage Accuracy in a Complex Dynamic Environment? A Study on $τ$-bench", + "url": "https://arxiv.org/abs/2508.20931", + "published": "2025-08-28", + "updated": "2025-09-01", + "authors": [ + "Venkatesh Mishra", + "Amir Saeidi", + "Satyam Raj", + "Mutsumi Nakamura", + "Jayanth Srinivasa", + "Gaowen Liu", + "Ali Payani", + "Chitta Baral" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 11, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.20931", + "source": "arxiv", + "source_id": "arxiv:2508.20931", + "pdf_url": "https://arxiv.org/pdf/2508.20931", + "primary_query": "function-calling" + }, + { + "id": "2607.05999", + "title": "AgoraSim: A Hybrid Agent-Based Modeling Framework", + "url": "https://arxiv.org/abs/2607.05999", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Chung-Chi Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.05999", + "source": "arxiv", + "source_id": "arxiv:2607.05999", + "pdf_url": "https://arxiv.org/pdf/2607.05999", + "primary_query": "llm-agent" + }, + { + "id": "2607.05975", + "title": "MCP-Enabled Agentic AI for Autonomous IPoDWDM Network Lifecycle Automation", + "url": "https://arxiv.org/abs/2607.05975", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Chunmin Xia", + "Jakub Harbaczewski", + "Nikhil Dsilva", + "Julie Raulin", + "Dominic Schneider", + "Achim Autenrieth" + ], + "categories": [ + "cs.NI", + "cs.AI", + "cs.MA", + "eess.SY" + ], + "topics": [ + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.05975", + "source": "arxiv", + "source_id": "arxiv:2607.05975", + "pdf_url": "https://arxiv.org/pdf/2607.05975", + "primary_query": "agentic-ai" + }, + { + "id": "2607.05958", + "title": "Agentic AI for IPoDWDM Network Lifecycle Automation: An MCP-Enabled Architecture", + "url": "https://arxiv.org/abs/2607.05958", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Chunmin Xia", + "Jakub Harbaczewski", + "Nikhil Dsilva", + "Julie Raulin", + "Dominic Schneider", + "Achim Autenrieth" + ], + "categories": [ + "cs.NI", + "cs.AI", + "eess.SP", + "eess.SY" + ], + "topics": [ + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.05958", + "source": "arxiv", + "source_id": "arxiv:2607.05958", + "pdf_url": "https://arxiv.org/pdf/2607.05958", + "primary_query": "agentic-ai" + }, + { + "id": "2607.06155", + "title": "When Does Tool Use Increase the Expressive Power of Finite-Precision Recurrent Models?", + "url": "https://arxiv.org/abs/2607.06155", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Nikola Zubić", + "Qian Li", + "Yuyi Wang", + "Davide Scaramuzza" + ], + "categories": [ + "cs.FL", + "cs.CC", + "cs.CL" + ], + "topics": [ + "memory", + "tool-use", + "world-model" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.06155", + "source": "arxiv", + "source_id": "arxiv:2607.06155", + "pdf_url": "https://arxiv.org/pdf/2607.06155", + "primary_query": "tool-use" + }, + { + "id": "2607.05762", + "title": "Articulating Assumptions in AI-Generated Scientific Analyses through Task Decomposition", + "url": "https://arxiv.org/abs/2607.05762", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Ahmed Hammad", + "Mihoko Nojiri" + ], + "categories": [ + "cs.SE", + "hep-ex", + "hep-ph" + ], + "topics": [ + "computer-use", + "multi-agent", + "planning", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.05762", + "source": "arxiv", + "source_id": "arxiv:2607.05762", + "pdf_url": "https://arxiv.org/pdf/2607.05762", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.04574", + "title": "A Few Teacher Steps Go a Long Way: Cost-Efficient On-Policy Data Augmentation for Agent Post-Training", + "url": "https://arxiv.org/abs/2607.04574", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Junze Ye", + "Jiayi Cheng", + "Miao Lu", + "Michal Mankowski", + "Jose Blanchet", + "Mohsen Bayati" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04574", + "source": "arxiv", + "source_id": "arxiv:2607.04574", + "pdf_url": "https://arxiv.org/pdf/2607.04574", + "primary_query": "llm-agent" + }, + { + "id": "2607.05277", + "title": "Untrusted Content Masking for Web Agents with Security Guarantees", + "url": "https://arxiv.org/abs/2607.05277", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Kristina Nikolić", + "Egor Zverev", + "Javier Rando", + "Matthew Jagielski", + "Edoardo Debenedetti", + "Florian Tramèr" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-safety", + "computer-use", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "tool-use", + "web-gui-agent" + ], + "arxiv_id": "2607.05277", + "source": "arxiv", + "source_id": "arxiv:2607.05277", + "pdf_url": "https://arxiv.org/pdf/2607.05277", + "primary_query": "tool-use" + }, + { + "id": "2607.04613", + "title": "Governed Individuation: Cryptographically Decoupling an Agent's Learning from Its Authority", + "url": "https://arxiv.org/abs/2607.04613", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Xue Qin", + "Simin Luan", + "Cong Yang", + "Zhijun Li" + ], + "categories": [ + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.04613", + "source": "arxiv", + "source_id": "arxiv:2607.04613", + "pdf_url": "https://arxiv.org/pdf/2607.04613", + "primary_query": "tool-use" + }, + { + "id": "2607.04371", + "title": "Nemotron-Labs-3-Puzzle-75B-A9B: Compressing Hybrid MoE LLMs", + "url": "https://arxiv.org/abs/2607.04371", + "published": "2026-07-05", + "updated": "2026-07-07", + "authors": [ + "Akhiad Bercovich", + "Talor Abramovich", + "Daniel Afrimi", + "Shay Aharon", + "Nir Ailon", + "Vladimir Anisimov", + "Omer Ullman Argov", + "Maor Ashkenazi", + "Tomer Asida", + "Nave Assaf", + "Tomer Bar Natan", + "Alexander Bukharin", + "Grzegorz Chlebus", + "Marcin Chochowski", + "Eric Chung", + "Mohammad Dabbah", + "Carlo del Mundo", + "Ewa Dobrowolska", + "Ido Galil", + "Yaniv Galron", + "Amnon Geifman", + "Yonatan Geifman", + "Izik Golan", + "Alex Gronskiy", + "Tomasz Grzegorzek", + "Netanel Haber", + "Lior Kadoch", + "Grzegorz Karch", + "Tomer Keren", + "Abhinav Khattar", + "Amir Klein", + "Tugrul Konuk", + "Roi Koren", + "Daniel Korzekwa", + "Shaun Kotek", + "Konstantinos Krommydas", + "Itay Levy", + "Ofri Masad", + "Yoav Miron", + "Pavlo Molchanov", + "Shahar Mor", + "Zach Moshe", + "Saurav Muralidharan", + "Najeeb Nabwani", + "Besmira Nushi", + "Mostofa Patwary", + "Omri Puny", + "Johannes Rausch", + "Tomer Ronen", + "Sepehr Sameni", + "Itamar Schen", + "Elad Segal", + "Daniel Serebrenik", + "Ido Shahaf", + "Soumye Singhal", + "Daniil Sorokin", + "Sharath Turuvekere Sreenivas", + "Marta Stepniewska-Dziubinska", + "Ali Taghibakhshi", + "Nima Tajbakhsh", + "Oren Tropp", + "Dor Tzur", + "Anna Warno", + "Yi-Fu Wu", + "Michal Zawalski", + "Jiaqi Zeng", + "Yian Zhang", + "Ran Zilberstein", + "Amit Zuker", + "Ran El-Yaniv" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2607.04371", + "source": "arxiv", + "source_id": "arxiv:2607.04371", + "pdf_url": "https://arxiv.org/pdf/2607.04371", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.03238", + "title": "An Empirical Study of Downstream Adaptation for Agent Skills", + "url": "https://arxiv.org/abs/2607.03238", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Xinjian Wu", + "Jingzhi Gong", + "Gunel Jahangirova", + "Zhenpeng Chen", + "Jie M. Zhang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.03238", + "source": "arxiv", + "source_id": "arxiv:2607.03238", + "pdf_url": "https://arxiv.org/pdf/2607.03238", + "primary_query": "llm-agent" + }, + { + "id": "2607.03048", + "title": "Compression, structure, and executor capability: a controlled real-cost decomposition of language-model agent skill optimisation", + "url": "https://arxiv.org/abs/2607.03048", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Xiaonan Xu", + "Wenjing Wu" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "computer-use", + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.03048", + "source": "arxiv", + "source_id": "arxiv:2607.03048", + "pdf_url": "https://arxiv.org/pdf/2607.03048", + "primary_query": "tool-use" + }, + { + "id": "2607.02873", + "title": "Determinants and Limits of LLM Security-Tool Orchestration: A Study with HexStrike-AI", + "url": "https://arxiv.org/abs/2607.02873", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Romain Gerard", + "Assmaa Zeghaider", + "Yan Guo" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.02873", + "source": "arxiv", + "source_id": "arxiv:2607.02873", + "pdf_url": "https://arxiv.org/pdf/2607.02873", + "primary_query": "tool-use" + }, + { + "id": "2607.03028", + "title": "HETERQA: Benchmarking Record Retrieval over Multiple Heterogeneous Sources", + "url": "https://arxiv.org/abs/2607.03028", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Yaodong Su", + "Hanchang Li", + "Quanqing Xu", + "Chuanhui Yang", + "Yixiang Fang" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.03028", + "source": "arxiv", + "source_id": "arxiv:2607.03028", + "pdf_url": "https://arxiv.org/pdf/2607.03028", + "primary_query": "rag-agent" + }, + { + "id": "2607.02459", + "title": "Language Models as Measurement Apparatus for Culture", + "url": "https://arxiv.org/abs/2607.02459", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Kent K. Chang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02459", + "source": "arxiv", + "source_id": "arxiv:2607.02459", + "pdf_url": "https://arxiv.org/pdf/2607.02459", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02376", + "title": "Hardware-Enforced Semantic Coordination for Safety-Critical Real-Time Autonomous Systems", + "url": "https://arxiv.org/abs/2607.02376", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Uwe M. Borghoff", + "Paolo Bottoni", + "Remo Pareschi" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use", + "world-model" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02376", + "source": "arxiv", + "source_id": "arxiv:2607.02376", + "pdf_url": "https://arxiv.org/pdf/2607.02376", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02370", + "title": "Understanding Agent-Based Patching of Compiler Missed Optimizations", + "url": "https://arxiv.org/abs/2607.02370", + "published": "2026-07-02", + "updated": "2026-07-03", + "authors": [ + "Batu Guan", + "Zirui Wang", + "Shaohua Li" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02370", + "source": "arxiv", + "source_id": "arxiv:2607.02370", + "pdf_url": "https://arxiv.org/pdf/2607.02370", + "primary_query": "coding-agent" + }, + { + "id": "2607.02134", + "title": "Coding-agents can replicate scientific machine learning papers", + "url": "https://arxiv.org/abs/2607.02134", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Atharva Hans", + "Ilias Bilionis" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02134", + "source": "arxiv", + "source_id": "arxiv:2607.02134", + "pdf_url": "https://arxiv.org/pdf/2607.02134", + "primary_query": "coding-agent" + }, + { + "id": "2607.01597", + "title": "A Single Patch Is Not Enough: Deterministic Fusion of Repair Candidates", + "url": "https://arxiv.org/abs/2607.01597", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Boyang Yang", + "Xiangliang Hu", + "Luyao Ren", + "Yanjun Chen", + "Bach Le", + "Tegawendé F. Bissyandé", + "Haoye Tian" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.01597", + "source": "arxiv", + "source_id": "arxiv:2607.01597", + "pdf_url": "https://arxiv.org/pdf/2607.01597", + "primary_query": "coding-agent" + }, + { + "id": "2607.01557", + "title": "DiPS: Dialogue Policy Selection for High-Stakes Persuasion Agents", + "url": "https://arxiv.org/abs/2607.01557", + "published": "2026-07-02", + "updated": "2026-07-03", + "authors": [ + "Tianyi Zhang", + "Mousumi Das", + "Abrar Anwar", + "Jesse Thomason", + "David Traum" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.01557", + "source": "arxiv", + "source_id": "arxiv:2607.01557", + "pdf_url": "https://arxiv.org/pdf/2607.01557", + "primary_query": "rag-agent" + }, + { + "id": "2607.01456", + "title": "From Anatomy to Smells: An Empirical Study of SKILL.md in Agent Skills", + "url": "https://arxiv.org/abs/2607.01456", + "published": "2026-07-01", + "updated": "2026-07-03", + "authors": [ + "David Boram Hong", + "Aaron Imani", + "Iftekhar Ahmed" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "computer-use", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01456", + "source": "arxiv", + "source_id": "arxiv:2607.01456", + "pdf_url": "https://arxiv.org/pdf/2607.01456", + "primary_query": "llm-agent" + }, + { + "id": "2607.00394", + "title": "When Classic Cache Policies Fail: Learning-Augmented Replacement for Semantic Retrieval Buffers", + "url": "https://arxiv.org/abs/2607.00394", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Yushi Sun", + "Bowen Cao", + "Wai Lam" + ], + "categories": [ + "cs.DB", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00394", + "source": "arxiv", + "source_id": "arxiv:2607.00394", + "pdf_url": "https://arxiv.org/pdf/2607.00394", + "primary_query": "llm-agent" + }, + { + "id": "2607.01507", + "title": "The Agentic Garden of Forking Paths", + "url": "https://arxiv.org/abs/2607.01507", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Jiacheng Miao", + "Jonathan K Pritchard", + "James Zou" + ], + "categories": [ + "cs.AI", + "stat.ME" + ], + "topics": [ + "agent-evaluation" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.01507", + "source": "arxiv", + "source_id": "arxiv:2607.01507", + "pdf_url": "https://arxiv.org/pdf/2607.01507", + "primary_query": "ai-agent" + }, + { + "id": "2607.00751", + "title": "SessionBound: Turning Enterprise Task Approval into Budgeted Database Sessions", + "url": "https://arxiv.org/abs/2607.00751", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Minmin Wu" + ], + "categories": [ + "cs.DB", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "planning", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.00751", + "source": "arxiv", + "source_id": "arxiv:2607.00751", + "pdf_url": "https://arxiv.org/pdf/2607.00751", + "primary_query": "ai-agent" + }, + { + "id": "2607.01465", + "title": "Beyond Next-Token Prediction: An RLVR Proof of Concept for Tool-Use Agents on Atlassian Workflows", + "url": "https://arxiv.org/abs/2607.01465", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Karthikeya Aditya Vissa", + "Sankalp Mane", + "Ananya Mantravadi", + "Harshit Rajgarhia", + "Abhishek Mukherji" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "rag", + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.01465", + "source": "arxiv", + "source_id": "arxiv:2607.01465", + "pdf_url": "https://arxiv.org/pdf/2607.01465", + "primary_query": "tool-use" + }, + { + "id": "2607.00871", + "title": "Self-Evolving Agents with Anytime-Valid Certificates", + "url": "https://arxiv.org/abs/2607.00871", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Biswa Sengupta" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.00871", + "source": "arxiv", + "source_id": "arxiv:2607.00871", + "pdf_url": "https://arxiv.org/pdf/2607.00871", + "primary_query": "coding-agent" + }, + { + "id": "2607.01044", + "title": "Robots Ask the Way: Communication-Enabled Social Navigation", + "url": "https://arxiv.org/abs/2607.01044", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Valentino Sacco", + "Luca Scofano", + "Indro Spinelli", + "Fabio Galasso" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.01044", + "source": "arxiv", + "source_id": "arxiv:2607.01044", + "pdf_url": "https://arxiv.org/pdf/2607.01044", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00245", + "title": "Agent-to-Agent Finance: Blockchain Payments and Trust Infrastructure for Autonomous AI Agents", + "url": "https://arxiv.org/abs/2607.00245", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Hui Gong" + ], + "categories": [ + "q-fin.GN" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.00245", + "source": "arxiv", + "source_id": "arxiv:2607.00245", + "pdf_url": "https://arxiv.org/pdf/2607.00245", + "primary_query": "ai-agent" + }, + { + "id": "2606.31272", + "title": "The Decomposition Is the Fingerprint: Per-Component Identity for Agent Skills", + "url": "https://arxiv.org/abs/2606.31272", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Hongliang Liu", + "Yuhao Wu", + "Tung-Ling Li" + ], + "categories": [ + "cs.CR", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.31272", + "source": "arxiv", + "source_id": "arxiv:2606.31272", + "pdf_url": "https://arxiv.org/pdf/2606.31272", + "primary_query": "ai-agent" + }, + { + "id": "2606.31036", + "title": "Teaching LLMs to Recommend and Defer in Underrepresented Epilepsy Care", + "url": "https://arxiv.org/abs/2606.31036", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Shreyas Rajesh", + "Kartik Sharma", + "Tonmoy Monsoor", + "Mehmet Yigit Turali", + "Richard Idro", + "Juliana Kayaga", + "Robert Sebunya", + "Tracy Tushabe Namata", + "Jessica Nichole Pasqua", + "Vwani Roychowdhury", + "Rajarshi Mazumder" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "rag" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31036", + "source": "arxiv", + "source_id": "arxiv:2606.31036", + "pdf_url": "https://arxiv.org/pdf/2606.31036", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.30801", + "title": "Using AI Agents to Automate Black-Box Audits of Personalization Algorithms at Scale", + "url": "https://arxiv.org/abs/2606.30801", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Alessandro Morosini", + "Sarah H. Cen", + "Andrew Ilyas", + "Hedi Driss", + "Aleksander Mądry", + "Chara Podimata" + ], + "categories": [ + "cs.CL", + "cs.CY", + "cs.LG", + "cs.SI" + ], + "topics": [ + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.30801", + "source": "arxiv", + "source_id": "arxiv:2606.30801", + "pdf_url": "https://arxiv.org/pdf/2606.30801", + "primary_query": "ai-agent" + }, + { + "id": "2606.30182", + "title": "MirrorCode: AI can rebuild entire programs from behavior alone", + "url": "https://arxiv.org/abs/2606.30182", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Tom Adamczewski", + "David Owen", + "David Rein", + "Florian Brand", + "Giles Edkins", + "Allen Hart", + "Daniel O'Connell" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.30182", + "source": "arxiv", + "source_id": "arxiv:2606.30182", + "pdf_url": "https://arxiv.org/pdf/2606.30182", + "primary_query": "ai-agent" + }, + { + "id": "2606.30911", + "title": "Why Solve It Twice? Hierarchical Accumulation of Skills for Transfer-Efficient ML Engineering", + "url": "https://arxiv.org/abs/2606.30911", + "published": "2026-06-29", + "updated": "2026-07-01", + "authors": [ + "Yongbin Kim", + "Yashar Talebirad", + "Osmar R. Zaiane" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30911", + "source": "arxiv", + "source_id": "arxiv:2606.30911", + "pdf_url": "https://arxiv.org/pdf/2606.30911", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29279", + "title": "Manufactured Confidence: How Memory Consolidation Turns Hearsay into Confident Facts", + "url": "https://arxiv.org/abs/2606.29279", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Alex Kwon" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "memory" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29279", + "source": "arxiv", + "source_id": "arxiv:2606.29279", + "pdf_url": "https://arxiv.org/pdf/2606.29279", + "primary_query": "llm-agent" + }, + { + "id": "2606.29194", + "title": "AI Trading's Alpha Singularity: Emergent Market Reasoning through Agent-to-Agent Self-Evolution", + "url": "https://arxiv.org/abs/2606.29194", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Yuqi Li", + "Siyuan Liu", + "Bingjun Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29194", + "source": "arxiv", + "source_id": "arxiv:2606.29194", + "pdf_url": "https://arxiv.org/pdf/2606.29194", + "primary_query": "llm-agent" + }, + { + "id": "2606.29280", + "title": "Deterministic Decisions for High-Stakes AI. A Zero-Egress Pipeline with the Deployability of RAG and the Accuracy of Machine Learning", + "url": "https://arxiv.org/abs/2606.29280", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Craig Atkinson" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.29280", + "source": "arxiv", + "source_id": "arxiv:2606.29280", + "pdf_url": "https://arxiv.org/pdf/2606.29280", + "primary_query": "rag-agent" + }, + { + "id": "2606.29113", + "title": "LLM Semantic Signaling Game and Mechanism Design: Systematic Blindness, Awareness Shaping, and Mindset Dynamics", + "url": "https://arxiv.org/abs/2606.29113", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Quanyan Zhu" + ], + "categories": [ + "cs.GT", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.29113", + "source": "arxiv", + "source_id": "arxiv:2606.29113", + "pdf_url": "https://arxiv.org/pdf/2606.29113", + "primary_query": "agentic-ai" + }, + { + "id": "2606.27845", + "title": "LLM Agents as Static Level-k Players in Behavioural Games", + "url": "https://arxiv.org/abs/2606.27845", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Po Han Teo" + ], + "categories": [ + "econ.GN", + "econ.TH" + ], + "topics": [ + "agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.27845", + "source": "arxiv", + "source_id": "arxiv:2606.27845", + "pdf_url": "https://arxiv.org/pdf/2606.27845", + "primary_query": "llm-agent" + }, + { + "id": "2606.27974", + "title": "ProMSA:Progressive Multimodal Search Agents for Knowledge-Based Visual Question Answering", + "url": "https://arxiv.org/abs/2606.27974", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "ZhengXian Wu", + "Hangrui Xu", + "Kai Shi", + "Zhuohong Chen", + "Yunyao Yu", + "Chuanrui Zhang", + "Zirui Liao", + "Jun Yang", + "Zhenyu Yang", + "Haonan Lu", + "Haoqian Wang" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent", + "tool-use" + ], + "arxiv_id": "2606.27974", + "source": "arxiv", + "source_id": "arxiv:2606.27974", + "pdf_url": "https://arxiv.org/pdf/2606.27974", + "primary_query": "rag-agent" + }, + { + "id": "2606.26959", + "title": "The Shift to Agentic AI: Evidence from Codex", + "url": "https://arxiv.org/abs/2606.26959", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Drew Johnston", + "David Holtz", + "Alex Martin Richmond", + "Christopher Ong", + "Prasanna Tambe", + "Aaron Chatterji" + ], + "categories": [ + "econ.GN" + ], + "topics": [ + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.26959", + "source": "arxiv", + "source_id": "arxiv:2606.26959", + "pdf_url": "https://arxiv.org/pdf/2606.26959", + "primary_query": "agentic-ai" + }, + { + "id": "2606.25605", + "title": "Constraint Tax in Open-Weight LLMs: An Empirical Study of Tool Calling Suppression Under Structured Output Constraints", + "url": "https://arxiv.org/abs/2606.25605", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Fangzheng Li", + "Aimin Zhang", + "Chen Lv" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.25605", + "source": "arxiv", + "source_id": "arxiv:2606.25605", + "pdf_url": "https://arxiv.org/pdf/2606.25605", + "primary_query": "tool-use" + }, + { + "id": "2606.24424", + "title": "Explainable AI for Next-Generation Wireless Physical Layer: Basics, State-of-the-Art, and Open Challenges", + "url": "https://arxiv.org/abs/2606.24424", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Bingnan Xiao", + "Shuyan Hu", + "Xiaojing Chen", + "Zhiyuan Zhai", + "Bingcong Li", + "Wei Ni", + "Xin Wang", + "Ekram Hossain" + ], + "categories": [ + "eess.SP" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.24424", + "source": "arxiv", + "source_id": "arxiv:2606.24424", + "pdf_url": "https://arxiv.org/pdf/2606.24424", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24470", + "title": "The Latent Bridge: A Continuous Slow-Fast Channel for Real-Time Game Agents", + "url": "https://arxiv.org/abs/2606.24470", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Bojie Li", + "Noah Shi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.24470", + "source": "arxiv", + "source_id": "arxiv:2606.24470", + "pdf_url": "https://arxiv.org/pdf/2606.24470", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.22813", + "title": "Active Inference as the Test-Time Scaling Law for Physical AI Agents", + "url": "https://arxiv.org/abs/2606.22813", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Omar Hashash", + "Christo Kurisummoottil Thomas", + "Walid Saad", + "Merouane Debbah", + "Karl Friston", + "Adeel Razi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "reasoning", + "world-model" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.22813", + "source": "arxiv", + "source_id": "arxiv:2606.22813", + "pdf_url": "https://arxiv.org/pdf/2606.22813", + "primary_query": "ai-agent" + }, + { + "id": "2606.23138", + "title": "Rising From the Ashes: How Agentic AI is Unblocking Challenges in Cybersecurity", + "url": "https://arxiv.org/abs/2606.23138", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Gabriela F. Ciocarlie", + "Kathrin Grosse", + "Somesh Jha", + "Daryna Oliynyk", + "Andrew Paverd", + "Christian Wressnegger" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-safety", + "reasoning" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.23138", + "source": "arxiv", + "source_id": "arxiv:2606.23138", + "pdf_url": "https://arxiv.org/pdf/2606.23138", + "primary_query": "agentic-ai" + }, + { + "id": "2606.23175", + "title": "Position: Correct Answer, Wrong Mechanism -- When AI Scientists Defend General Claims Their Own Data Contradicts", + "url": "https://arxiv.org/abs/2606.23175", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Steven Young Eulig" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "world-model" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.23175", + "source": "arxiv", + "source_id": "arxiv:2606.23175", + "pdf_url": "https://arxiv.org/pdf/2606.23175", + "primary_query": "coding-agent" + }, + { + "id": "2606.22906", + "title": "From Fragments to Paths: Task-Level Context Recovery for Large Industrial Codebases", + "url": "https://arxiv.org/abs/2606.22906", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Jiawei He", + "Weisong Sun", + "Mengyu Shi", + "Jie Jia", + "Tong Bian", + "Xikai Yang", + "Dong Sun" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.22906", + "source": "arxiv", + "source_id": "arxiv:2606.22906", + "pdf_url": "https://arxiv.org/pdf/2606.22906", + "primary_query": "coding-agent" + }, + { + "id": "2606.23189", + "title": "Capable but Careless: Do Computer-Use Agents Follow Contextual Integrity?", + "url": "https://arxiv.org/abs/2606.23189", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Anmol Goel", + "Iryna Gurevych" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.23189", + "source": "arxiv", + "source_id": "arxiv:2606.23189", + "pdf_url": "https://arxiv.org/pdf/2606.23189", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.21959", + "title": "OpenBioRQ: Unsolved Biomedical Research Questions for Agents", + "url": "https://arxiv.org/abs/2606.21959", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Minbyul Jeong" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.21959", + "source": "arxiv", + "source_id": "arxiv:2606.21959", + "pdf_url": "https://arxiv.org/pdf/2606.21959", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.21151", + "title": "Context-Aware Generative AI for Automated Telecom Test Script Generation", + "url": "https://arxiv.org/abs/2606.21151", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Gautam Prasad", + "Chandramohan T. N.", + "Joy Bose" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.NI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2606.21151", + "source": "arxiv", + "source_id": "arxiv:2606.21151", + "pdf_url": "https://arxiv.org/pdf/2606.21151", + "primary_query": "ai-agent" + }, + { + "id": "2606.21037", + "title": "Honeyquest for LLMs: Rethinking Cyber Deception for AI Attackers", + "url": "https://arxiv.org/abs/2606.21037", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Kerri Prinos", + "Lilianne Brush", + "Cameron Denton" + ], + "categories": [ + "cs.CR", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.21037", + "source": "arxiv", + "source_id": "arxiv:2606.21037", + "pdf_url": "https://arxiv.org/pdf/2606.21037", + "primary_query": "ai-agent" + }, + { + "id": "2606.21804", + "title": "Is Agent Code Less Maintainable Than Human Code?", + "url": "https://arxiv.org/abs/2606.21804", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Shaswat Patel", + "Betty Li Hou", + "Arun Purohit", + "Kai Xu", + "Jane Pan", + "He He", + "Valerie Chen" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.21804", + "source": "arxiv", + "source_id": "arxiv:2606.21804", + "pdf_url": "https://arxiv.org/pdf/2606.21804", + "primary_query": "coding-agent" + }, + { + "id": "2606.20002", + "title": "Connect the Dots: Training LLMs for Long-Lifecycle Agents with Cross-Domain Generalization Via Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.20002", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Yanxi Chen", + "Weijie Shi", + "Yuexiang Xie", + "Boyi Hu", + "Yaliang Li", + "Bolin Ding", + "Jingren Zhou" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.20002", + "source": "arxiv", + "source_id": "arxiv:2606.20002", + "pdf_url": "https://arxiv.org/pdf/2606.20002", + "primary_query": "ai-agent" + }, + { + "id": "2606.20158", + "title": "N-Version Programming with Coding Agents", + "url": "https://arxiv.org/abs/2606.20158", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Javier Ron", + "Benoit Baudry", + "Martin Monperrus" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.20158", + "source": "arxiv", + "source_id": "arxiv:2606.20158", + "pdf_url": "https://arxiv.org/pdf/2606.20158", + "primary_query": "coding-agent" + }, + { + "id": "2606.19830", + "title": "JAMER: Project-Level Code Framework Dataset and Benchmark on Professional Game Engines", + "url": "https://arxiv.org/abs/2606.19830", + "published": "2026-06-18", + "updated": "2026-06-21", + "authors": [ + "Jianwen Sun", + "Chuanhao Li", + "Zizhen Li", + "Yukang Feng", + "Fanrui Zhang", + "Yifei Huang", + "Yu Dai", + "Kaipeng Zhang" + ], + "categories": [ + "cs.SE", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.19830", + "source": "arxiv", + "source_id": "arxiv:2606.19830", + "pdf_url": "https://arxiv.org/pdf/2606.19830", + "primary_query": "coding-agent" + }, + { + "id": "2606.20363", + "title": "Automating SKILL.md Generation for Computer-Using Agents via Interaction Trajectory Mining", + "url": "https://arxiv.org/abs/2606.20363", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Yuexing Hao", + "Xiaomin Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20363", + "source": "arxiv", + "source_id": "arxiv:2606.20363", + "pdf_url": "https://arxiv.org/pdf/2606.20363", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.20537", + "title": "Execution-State Capsules: Graph-Bound Execution-State Checkpoint and Restore for Low-Latency, Small-Batch, On-Device Physical-AI Serving", + "url": "https://arxiv.org/abs/2606.20537", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Liang Su" + ], + "categories": [ + "cs.LG", + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "rag" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.20537", + "source": "arxiv", + "source_id": "arxiv:2606.20537", + "pdf_url": "https://arxiv.org/pdf/2606.20537", + "primary_query": "planning-agent" + }, + { + "id": "2606.19116", + "title": "Towards an Agent-First Web: Redesigning the Web for AI Agents", + "url": "https://arxiv.org/abs/2606.19116", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Eranga Bandara", + "Ross Gore", + "Ravi Mukkamala", + "Asanga Gunaratna", + "Safdar H. Bouk", + "Xueping Liang", + "Peter Foytik", + "Abdul Rahman", + "Sachini Rajapakse", + "Isurunima Kularathna", + "Pramoda Karunarathna", + "Chalani Rajapakse", + "Ng Wee Keong", + "Kasun De Zoysa", + "Tharaka Hewa", + "Amin Hass", + "Wathsala Herath", + "Aruna Withanage", + "Nilaan Loganathan", + "Atmaram Yarlagadda", + "Sachin Shetty" + ], + "categories": [ + "cs.AI", + "cs.CY" + ], + "topics": [ + "computer-use", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.19116", + "source": "arxiv", + "source_id": "arxiv:2606.19116", + "pdf_url": "https://arxiv.org/pdf/2606.19116", + "primary_query": "ai-agent" + }, + { + "id": "2606.18716", + "title": "Human-AI Agent Interaction in a Business Context", + "url": "https://arxiv.org/abs/2606.18716", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Kathrin Paimann", + "Elizangela Valarini", + "Sebastian Juhl" + ], + "categories": [ + "cs.HC", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.18716", + "source": "arxiv", + "source_id": "arxiv:2606.18716", + "pdf_url": "https://arxiv.org/pdf/2606.18716", + "primary_query": "ai-agent" + }, + { + "id": "2606.18831", + "title": "Beyond Reward Engineering: A Data Recipe for Long-Context Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.18831", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Xiaoyue Xu", + "Sikui Zhang", + "Xiaorong Wang", + "Xu Han", + "Chaojun Xiao" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.18831", + "source": "arxiv", + "source_id": "arxiv:2606.18831", + "pdf_url": "https://arxiv.org/pdf/2606.18831", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.17645", + "title": "Beyond Domains: Reusing Web Skills via Transferable Interaction Patterns", + "url": "https://arxiv.org/abs/2606.17645", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Shiqi He", + "Yue Cui", + "Feijie Wu", + "Xinyu Ma", + "Jiaheng Lu", + "Yaliang Li", + "Bolin Ding", + "Mosharaf Chowdhury" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.17645", + "source": "arxiv", + "source_id": "arxiv:2606.17645", + "pdf_url": "https://arxiv.org/pdf/2606.17645", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.28370", + "title": "Conversational Query Engine for Mixed-Modality Heterogeneous Enterprise Data Sources", + "url": "https://arxiv.org/abs/2606.28370", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Darshita Rathore", + "Vineet Kumar", + "Vaibhav Singal", + "Ankur Vivek Singh", + "Anindya Moitra" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.28370", + "source": "arxiv", + "source_id": "arxiv:2606.28370", + "pdf_url": "https://arxiv.org/pdf/2606.28370", + "primary_query": "rag-agent" + }, + { + "id": "2606.12908", + "title": "SENTINEL: Failure-Driven Reinforcement Learning for Training Tool-Using Language Model Agents", + "url": "https://arxiv.org/abs/2606.12908", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Ziyi Wang", + "Yuxuan Lu", + "Yimeng Zhang", + "Qun Liu", + "Chen Luo", + "Jiri Gesi", + "Hanqing Lu", + "Yisi Sang", + "Manling Li", + "Jing Huang", + "Dakuo Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.12908", + "source": "arxiv", + "source_id": "arxiv:2606.12908", + "pdf_url": "https://arxiv.org/pdf/2606.12908", + "primary_query": "tool-use" + }, + { + "id": "2606.13949", + "title": "Minim: Privacy-Aware Minimal View for Agents via Trusted Local Sanitization", + "url": "https://arxiv.org/abs/2606.13949", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Hexuan Yu", + "Chaoyu Zhang", + "Heng Jin", + "Shanghao Shi", + "Ning Zhang", + "Y. Thomas Hou", + "Wenjing Lou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.13949", + "source": "arxiv", + "source_id": "arxiv:2606.13949", + "pdf_url": "https://arxiv.org/pdf/2606.13949", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.10299", + "title": "What Spatial Memory Must Store: Occlusion as the Test for Language-Agent Memory", + "url": "https://arxiv.org/abs/2606.10299", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Doeon Kwon", + "Junho Bang" + ], + "categories": [ + "cs.AI", + "cs.CV", + "cs.MA" + ], + "topics": [ + "memory" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agent-memory", + "language-agent" + ], + "arxiv_id": "2606.10299", + "source": "arxiv", + "source_id": "arxiv:2606.10299", + "pdf_url": "https://arxiv.org/pdf/2606.10299", + "primary_query": "agent-memory" + }, + { + "id": "2606.11350", + "title": "When More Documents Hurt RAG: Mitigating Vector Search Dilution with Domain-Scoped, Model-Agnostic Retrieval", + "url": "https://arxiv.org/abs/2606.11350", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Nabaraj Subedi", + "Ahmed Abdelaty", + "Shivanand Venkanna Sheshappanavar" + ], + "categories": [ + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.11350", + "source": "arxiv", + "source_id": "arxiv:2606.11350", + "pdf_url": "https://arxiv.org/pdf/2606.11350", + "primary_query": "rag-agent" + }, + { + "id": "2606.09935", + "title": "GitInject: Real-World Prompt Injection Attacks in AI-Powered CI/CD Pipelines", + "url": "https://arxiv.org/abs/2606.09935", + "published": "2026-06-07", + "updated": "2026-06-07", + "authors": [ + "Jafar Isbarov", + "Umid Suleymanov", + "Ilia Shumailov", + "Murat Kantarcioglu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.09935", + "source": "arxiv", + "source_id": "arxiv:2606.09935", + "pdf_url": "https://arxiv.org/pdf/2606.09935", + "primary_query": "agent-safety" + }, + { + "id": "2606.07017", + "title": "The Sim-to-Real Gap of Foundation Model Agents: A Unified MDP Perspective", + "url": "https://arxiv.org/abs/2606.07017", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Xiaoou Liu", + "Tiejin Chen", + "Weibo Li", + "Xiyang Hu", + "Hua Wei" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.ET" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.07017", + "source": "arxiv", + "source_id": "arxiv:2606.07017", + "pdf_url": "https://arxiv.org/pdf/2606.07017", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.03889", + "title": "RealClawBench: Live OpenClaw Benchmarks from Real Developer-Agent Sessions", + "url": "https://arxiv.org/abs/2606.03889", + "published": "2026-06-02", + "updated": "2026-06-05", + "authors": [ + "Zongwei Lv", + "Zhewen Tan", + "Yaoming Li", + "Yilun Yao", + "Yuxuan Tian", + "Lin Sun", + "Xiangzheng Zhang", + "Weihong Lin", + "Tong Yang", + "Guangxiang Zhao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.03889", + "source": "arxiv", + "source_id": "arxiv:2606.03889", + "pdf_url": "https://arxiv.org/pdf/2606.03889", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.01751", + "title": "SparseX: Efficient Segment-Level KV Cache Sharing for Interleaved LLM Serving", + "url": "https://arxiv.org/abs/2606.01751", + "published": "2026-06-01", + "updated": "2026-06-07", + "authors": [ + "Quqing Zhang", + "Kai Chen", + "Ning Liao", + "Zehao Lin", + "Bo Tang", + "Feiyu Xiong", + "Zhiyu Li", + "Xiaoxing Wang" + ], + "categories": [ + "cs.PF" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.01751", + "source": "arxiv", + "source_id": "arxiv:2606.01751", + "pdf_url": "https://arxiv.org/pdf/2606.01751", + "primary_query": "rag-agent" + }, + { + "id": "2606.02643", + "title": "Inference Cost Attacks for Retrieval-Augmented Large Language Models", + "url": "https://arxiv.org/abs/2606.02643", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Chengliang Liu", + "Liangbo Ning", + "Yujuan Ding", + "Wenqi Fan" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.DB" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.02643", + "source": "arxiv", + "source_id": "arxiv:2606.02643", + "pdf_url": "https://arxiv.org/pdf/2606.02643", + "primary_query": "rag-agent" + }, + { + "id": "2606.00750", + "title": "I-WebGenBench : Evaluating Interactivity in LLM-Generated Scientific Web Applications", + "url": "https://arxiv.org/abs/2606.00750", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Dasen Dai", + "Biao Wu", + "Meng Fang", + "Shuoqi Li", + "Wenhao Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.00750", + "source": "arxiv", + "source_id": "arxiv:2606.00750", + "pdf_url": "https://arxiv.org/pdf/2606.00750", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.29559", + "title": "LiteCoder-Terminal: Scaling Long-Horizon Terminal Environments for Learning Language Agents", + "url": "https://arxiv.org/abs/2605.29559", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Xiaoxuan Peng", + "Kaiqi Zhang", + "Xinyu Lu", + "Boxi Cao", + "Yaojie Lu", + "Hongyu Lin", + "Xianpei Han", + "Le Sun" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "planning", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.29559", + "source": "arxiv", + "source_id": "arxiv:2605.29559", + "pdf_url": "https://arxiv.org/pdf/2605.29559", + "primary_query": "language-agent" + }, + { + "id": "2605.29491", + "title": "The Curse of Helpfulness: Inverse Scaling Law in Robustness to Distractor Instructions via DistractionIF", + "url": "https://arxiv.org/abs/2605.29491", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Zeli Su", + "Zhankai Xu", + "Tianlei Chen", + "Longfei Zheng", + "Xiaolu Zhang", + "Jun Zhou", + "Wentao Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.29491", + "source": "arxiv", + "source_id": "arxiv:2605.29491", + "pdf_url": "https://arxiv.org/pdf/2605.29491", + "primary_query": "rag-agent" + }, + { + "id": "2605.26352", + "title": "RICE-PO: Turning Retrieval Interactions into Credit Signals for Reasoning Agents", + "url": "https://arxiv.org/abs/2605.26352", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Mingchen Li", + "Hansi Zeng", + "Zhuo Qian", + "Jiatan Huang", + "Hamed Zamani", + "Hong Yu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.26352", + "source": "arxiv", + "source_id": "arxiv:2605.26352", + "pdf_url": "https://arxiv.org/pdf/2605.26352", + "primary_query": "language-agent" + }, + { + "id": "2605.25988", + "title": "What Makes a Medical Checker Trainable? Diagnosing Signal Collapse and Reward Hacking in Checker-Guided RAG for Biomedical QA", + "url": "https://arxiv.org/abs/2605.25988", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Yuelyu Ji", + "Min Gu Kwak", + "Hang Zhang", + "Xizhi Wu", + "Chenyu Li", + "Yanshan Wan" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.25988", + "source": "arxiv", + "source_id": "arxiv:2605.25988", + "pdf_url": "https://arxiv.org/pdf/2605.25988", + "primary_query": "rag-agent" + }, + { + "id": "2605.23281", + "title": "DepthAgent: Towards Better Universal Depth Estimation via Sample-wise Expert Selection", + "url": "https://arxiv.org/abs/2605.23281", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Jie Zhu", + "Girish Chandar Ganesan", + "Xiaoming Liu" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.23281", + "source": "arxiv", + "source_id": "arxiv:2605.23281", + "pdf_url": "https://arxiv.org/pdf/2605.23281", + "primary_query": "language-agent" + }, + { + "id": "2605.20312", + "title": "Pramana: A Protocol-Layer Treatment of Claim Verification in Autonomous Agent Networks", + "url": "https://arxiv.org/abs/2605.20312", + "published": "2026-05-19", + "updated": "2026-05-19", + "authors": [ + "Ravi Kiran Kadaboina" + ], + "categories": [ + "cs.CR", + "cs.LO", + "cs.MA" + ], + "topics": [ + "rag", + "reasoning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.20312", + "source": "arxiv", + "source_id": "arxiv:2605.20312", + "pdf_url": "https://arxiv.org/pdf/2605.20312", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.13438", + "title": "CogniFold: Always-On Proactive Memory via Cognitive Folding", + "url": "https://arxiv.org/abs/2605.13438", + "published": "2026-05-13", + "updated": "2026-06-17", + "authors": [ + "Suli Wang", + "Yiqun Duan", + "Yu Deng", + "Rundong Zhao", + "Dai Shi", + "Minghua Deng", + "Chen Chen", + "Xinliang Zhou" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.13438", + "source": "arxiv", + "source_id": "arxiv:2605.13438", + "pdf_url": "https://arxiv.org/pdf/2605.13438", + "primary_query": "agent-memory" + }, + { + "id": "2605.11359", + "title": "CVEvolve: Autonomous Algorithm Discovery for Unstructured Scientific Data Processing", + "url": "https://arxiv.org/abs/2605.11359", + "published": "2026-05-12", + "updated": "2026-06-01", + "authors": [ + "Ming Du", + "Xiangyu Yin", + "Yanqi Luo", + "Dishant Beniwal", + "Songyuan Tang", + "Hemant Sharma", + "Mathew J. Cherukara" + ], + "categories": [ + "cs.AI", + "physics.data-an" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.11359", + "source": "arxiv", + "source_id": "arxiv:2605.11359", + "pdf_url": "https://arxiv.org/pdf/2605.11359", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.08721", + "title": "Breaking the Impasse: Dual-Scale Evolutionary Policy Training for Social Language Agents", + "url": "https://arxiv.org/abs/2605.08721", + "published": "2026-05-09", + "updated": "2026-05-09", + "authors": [ + "Minzheng Wang", + "Run Luo", + "Yanbo Wang", + "Zichen Liu", + "Yuqiao Tan", + "Tao Tan", + "Xu Nan", + "Yinhe Zheng", + "Wenji Mao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.08721", + "source": "arxiv", + "source_id": "arxiv:2605.08721", + "pdf_url": "https://arxiv.org/pdf/2605.08721", + "primary_query": "language-agent" + }, + { + "id": "2605.03159", + "title": "Learning Correct Behavior from Examples: Validating Sequential Execution in Autonomous Agents", + "url": "https://arxiv.org/abs/2605.03159", + "published": "2026-05-04", + "updated": "2026-05-04", + "authors": [ + "Reshabh K Sharma", + "Gaurav Mittal", + "Yu Hu" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "rag" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.03159", + "source": "arxiv", + "source_id": "arxiv:2605.03159", + "pdf_url": "https://arxiv.org/pdf/2605.03159", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.11839", + "title": "Beyond Static Sandboxing: Learned Capability Governance for Autonomous AI Agents", + "url": "https://arxiv.org/abs/2604.11839", + "published": "2026-04-12", + "updated": "2026-05-03", + "authors": [ + "Bronislav Sidik", + "Lior Rokach" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.11839", + "source": "arxiv", + "source_id": "arxiv:2604.11839", + "pdf_url": "https://arxiv.org/pdf/2604.11839", + "primary_query": "agent-safety" + }, + { + "id": "2603.27742", + "title": "TIR-Agent: Training an Explorative and Efficient Agent for Image Restoration", + "url": "https://arxiv.org/abs/2603.27742", + "published": "2026-03-29", + "updated": "2026-03-29", + "authors": [ + "Yisheng Zhang", + "Guoli Jia", + "Haote Hu", + "Shanxu Zhao", + "Kaikai Zhao", + "Long Sun", + "Xinwei Long", + "Kai Tian", + "Che Jiang", + "Zhaoxiang Liu", + "Kai Wang", + "Shiguo Lian", + "Kaiyan Zhang", + "Bowen Zhou" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.27742", + "source": "arxiv", + "source_id": "arxiv:2603.27742", + "pdf_url": "https://arxiv.org/pdf/2603.27742", + "primary_query": "language-agent" + }, + { + "id": "2603.05413", + "title": "Building Enterprise Realtime Voice Agents from Scratch: A Technical Tutorial", + "url": "https://arxiv.org/abs/2603.05413", + "published": "2026-03-05", + "updated": "2026-03-17", + "authors": [ + "Jielin Qiu", + "Zixiang Chen", + "Liangwei Yang", + "Ming Zhu", + "Zhiwei Liu", + "Juntao Tan", + "Wenting Zhao", + "Rithesh Murthy", + "Roshan Ram", + "Akshara Prabhakar", + "Shelby Heinecke", + "Caiming Xiong", + "Silvio Savarese", + "Huan Wang" + ], + "categories": [ + "cs.SD" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.05413", + "source": "arxiv", + "source_id": "arxiv:2603.05413", + "pdf_url": "https://arxiv.org/pdf/2603.05413", + "primary_query": "function-calling" + }, + { + "id": "2603.00991", + "title": "Tracking Capabilities for Safer Agents", + "url": "https://arxiv.org/abs/2603.00991", + "published": "2026-03-01", + "updated": "2026-05-07", + "authors": [ + "Martin Odersky", + "Yaoyu Zhao", + "Yichen Xu", + "Oliver Bračevac", + "Cao Nguyen Pham" + ], + "categories": [ + "cs.AI", + "cs.PL" + ], + "topics": [ + "agent-safety", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.00991", + "source": "arxiv", + "source_id": "arxiv:2603.00991", + "pdf_url": "https://arxiv.org/pdf/2603.00991", + "primary_query": "agent-safety" + }, + { + "id": "2602.15197", + "title": "OpaqueToolsBench: Learning Nuances of Tool Behavior Through Interaction", + "url": "https://arxiv.org/abs/2602.15197", + "published": "2026-02-16", + "updated": "2026-02-16", + "authors": [ + "Skyler Hallinan", + "Thejas Venkatesh", + "Xiang Ren", + "Sai Praneeth Karimireddy", + "Ashwin Paranjape", + "Yuhao Zhang", + "Jack Hessel" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2602.15197", + "source": "arxiv", + "source_id": "arxiv:2602.15197", + "pdf_url": "https://arxiv.org/pdf/2602.15197", + "primary_query": "function-calling" + }, + { + "id": "2602.03025", + "title": "RC-GRPO: Reward-Conditioned Group Relative Policy Optimization for Multi-Turn Tool Calling Agents", + "url": "https://arxiv.org/abs/2602.03025", + "published": "2026-02-03", + "updated": "2026-02-03", + "authors": [ + "Haitian Zhong", + "Jixiu Zhai", + "Lei Song", + "Jiang Bian", + "Qiang Liu", + "Tieniu Tan" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2602.03025", + "source": "arxiv", + "source_id": "arxiv:2602.03025", + "pdf_url": "https://arxiv.org/pdf/2602.03025", + "primary_query": "function-calling" + }, + { + "id": "2602.03022", + "title": "STAR: Similarity-guided Teacher-Assisted Refinement for Super-Tiny Function Calling Models", + "url": "https://arxiv.org/abs/2602.03022", + "published": "2026-02-03", + "updated": "2026-02-24", + "authors": [ + "Jiliang Ni", + "Jiachen Pu", + "Zhongyi Yang", + "Jingfeng Luo", + "Conggang Hu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2602.03022", + "source": "arxiv", + "source_id": "arxiv:2602.03022", + "pdf_url": "https://arxiv.org/pdf/2602.03022", + "primary_query": "function-calling" + }, + { + "id": "2601.17829", + "title": "Linguistic and Argument Diversity in Synthetic Data for Function-Calling Agents", + "url": "https://arxiv.org/abs/2601.17829", + "published": "2026-01-25", + "updated": "2026-01-25", + "authors": [ + "Dan Greenstein", + "Zohar Karnin", + "Chen Amiraz", + "Oren Somekh" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.17829", + "source": "arxiv", + "source_id": "arxiv:2601.17829", + "pdf_url": "https://arxiv.org/pdf/2601.17829", + "primary_query": "function-calling" + }, + { + "id": "2512.17052", + "title": "Dynamic Tool Dependency Retrieval for Lightweight Function Calling", + "url": "https://arxiv.org/abs/2512.17052", + "published": "2025-12-18", + "updated": "2026-04-17", + "authors": [ + "Bhrij Patel", + "Davide Belli", + "Amir Jalalirad", + "Maximilian Arnold", + "Aleksandr Ermolov", + "Bence Major" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.17052", + "source": "arxiv", + "source_id": "arxiv:2512.17052", + "pdf_url": "https://arxiv.org/pdf/2512.17052", + "primary_query": "function-calling" + }, + { + "id": "2510.10197", + "title": "Don't Just Fine-tune the Agent, Tune the Environment", + "url": "https://arxiv.org/abs/2510.10197", + "published": "2025-10-11", + "updated": "2026-01-30", + "authors": [ + "Siyuan Lu", + "Zechuan Wang", + "Hongxuan Zhang", + "Qintong Wu", + "Leilei Gan", + "Chenyi Zhuang", + "Jinjie Gu", + "Tao Lin" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.10197", + "source": "arxiv", + "source_id": "arxiv:2510.10197", + "pdf_url": "https://arxiv.org/pdf/2510.10197", + "primary_query": "function-calling" + }, + { + "id": "2510.06727", + "title": "Scaling LLM Multi-turn RL with End-to-end Summarization-based Context Management", + "url": "https://arxiv.org/abs/2510.06727", + "published": "2025-10-08", + "updated": "2025-10-08", + "authors": [ + "Miao Lu", + "Weiwei Sun", + "Weihua Du", + "Zhan Ling", + "Xuesong Yao", + "Kang Liu", + "Jiecao Chen" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use" + ], + "score": 10, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.06727", + "source": "arxiv", + "source_id": "arxiv:2510.06727", + "pdf_url": "https://arxiv.org/pdf/2510.06727", + "primary_query": "function-calling" + }, + { + "id": "2607.06126", + "title": "Causal Inference with Video Features as Treatments", + "url": "https://arxiv.org/abs/2607.06126", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Kentaro Nakamura", + "Adam Breuer", + "Michael H. Crespin", + "Bryce J. Dietrich", + "Kosuke Imai" + ], + "categories": [ + "stat.AP" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.06126", + "source": "arxiv", + "source_id": "arxiv:2607.06126", + "pdf_url": "https://arxiv.org/pdf/2607.06126", + "primary_query": "ai-agent" + }, + { + "id": "2607.06214", + "title": "A toy framework for single and multi-agent human-AI curiosity ecosystems", + "url": "https://arxiv.org/abs/2607.06214", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Ilya E. Monosov" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "multi-agent" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.06214", + "source": "arxiv", + "source_id": "arxiv:2607.06214", + "pdf_url": "https://arxiv.org/pdf/2607.06214", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04763", + "title": "Multi-Turn On-Policy Distillation with Prefix Replay", + "url": "https://arxiv.org/abs/2607.04763", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Baohao Liao", + "Hanze Dong", + "Christof Monz", + "Xinxing Xu", + "Li Dong", + "Furu Wei" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL", + "stat.ML" + ], + "topics": [ + "reasoning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04763", + "source": "arxiv", + "source_id": "arxiv:2607.04763", + "pdf_url": "https://arxiv.org/pdf/2607.04763", + "primary_query": "llm-agent" + }, + { + "id": "2607.04631", + "title": "Formal Disco: Scalable Open-Ended Generation of Formally Verified Programs", + "url": "https://arxiv.org/abs/2607.04631", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Gabriel Poesia", + "Simon Henniger", + "Tzu-Han Hsu", + "Yilun Du", + "Nada Amin" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.04631", + "source": "arxiv", + "source_id": "arxiv:2607.04631", + "pdf_url": "https://arxiv.org/pdf/2607.04631", + "primary_query": "ai-agent" + }, + { + "id": "2607.04708", + "title": "Strategic Buying Agents", + "url": "https://arxiv.org/abs/2607.04708", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Mingyang Fu", + "Ming Hu" + ], + "categories": [ + "econ.TH", + "cs.AI", + "cs.CY", + "cs.GT", + "cs.HC" + ], + "topics": [ + "agent-evaluation" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04708", + "source": "arxiv", + "source_id": "arxiv:2607.04708", + "pdf_url": "https://arxiv.org/pdf/2607.04708", + "primary_query": "agentic-ai" + }, + { + "id": "2607.05471", + "title": "KAT-Coder-V2.5 Technical Report", + "url": "https://arxiv.org/abs/2607.05471", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Bo Huang", + "Fengxiang Li", + "Hao Xu", + "Haoyang Huang", + "Hongyi Fu", + "Jinhua Hao", + "Kun Yuan", + "Minglei Zhang", + "Pengcheng Xu", + "Shiyang Liu", + "Wenhao Zhuang", + "Yuze Shi", + "Zongxian Feng", + "Chao Wang", + "Cheng He", + "Chongling Rao", + "Deyu Cao", + "Fan Yang", + "Gang Xiong", + "Haochen Liu", + "Jiabao Li", + "Jian Liang", + "Jinghui Jia", + "Jingwen Chang", + "Jun Du", + "Junyu Shi", + "Min Li", + "Mingqi Wu", + "Qiang Gao", + "Shangpeng Yan", + "Shaotong Qi", + "Shu Xu", + "Shuo Zhou", + "Tiankuo Xu", + "Tong Zheng", + "Weilun Zhao", + "Xiancheng Meng", + "Xianda Sun", + "Xiaoyu Jiang", + "Xunhao Jia", + "Yao Xia", + "Yimeng Xu", + "Yinghan Cui", + "Yingpeng Chen", + "Yiwen Ning", + "Yong Wang", + "Yuxuan Sun", + "Zhongsheng Liu", + "Ming Sun", + "Cheng Luo", + "Chen Yang", + "Han Li", + "Kun Gai" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation", + "tool-use" + ], + "arxiv_id": "2607.05471", + "source": "arxiv", + "source_id": "arxiv:2607.05471", + "pdf_url": "https://arxiv.org/pdf/2607.05471", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.05465", + "title": "CanvasAgent: Enabling Complex Image Creation and Editing via Visual Tool Orchestration", + "url": "https://arxiv.org/abs/2607.05465", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Hairui Zhu", + "Yiying Yang", + "Tengjin Weng", + "Ziyu Lu", + "Xiao Yao", + "Xiaoyang Ye", + "Lin Ma", + "Wenhao Jiang" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.05465", + "source": "arxiv", + "source_id": "arxiv:2607.05465", + "pdf_url": "https://arxiv.org/pdf/2607.05465", + "primary_query": "tool-use" + }, + { + "id": "2607.04508", + "title": "Compressing the Validation Bottleneck: An Agentic Self-Driving Lab for Scientific Discovery", + "url": "https://arxiv.org/abs/2607.04508", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Kyunghoon Hur", + "Chihun Lee" + ], + "categories": [ + "cs.AI", + "cs.RO" + ], + "topics": [ + "planning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04508", + "source": "arxiv", + "source_id": "arxiv:2607.04508", + "pdf_url": "https://arxiv.org/pdf/2607.04508", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02514", + "title": "Distributed Attacks in Persistent-State AI Control", + "url": "https://arxiv.org/abs/2607.02514", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Josh Hills", + "Ida Caspary", + "Asa Cooper Stickland" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02514", + "source": "arxiv", + "source_id": "arxiv:2607.02514", + "pdf_url": "https://arxiv.org/pdf/2607.02514", + "primary_query": "coding-agent" + }, + { + "id": "2607.02357", + "title": "Cloak and Detonate: Scanner Evasion and Dynamic Detection of Agent Skill Malware", + "url": "https://arxiv.org/abs/2607.02357", + "published": "2026-07-02", + "updated": "2026-07-03", + "authors": [ + "Zimo Ji", + "Congying Xu", + "Zongjie Li", + "Yudong Gao", + "Xin Wei", + "Shuai Wang", + "Shing-Chi Cheung" + ], + "categories": [ + "cs.CR", + "cs.SE" + ], + "topics": [ + "coding-agent" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02357", + "source": "arxiv", + "source_id": "arxiv:2607.02357", + "pdf_url": "https://arxiv.org/pdf/2607.02357", + "primary_query": "coding-agent" + }, + { + "id": "2607.02057", + "title": "Prompt Coverage Adequacy", + "url": "https://arxiv.org/abs/2607.02057", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Florian Tambon", + "Michael Konstantinou", + "Cedric Richter", + "Charles Chenouard", + "Mark Harman", + "Mike Papadakis" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2607.02057", + "source": "arxiv", + "source_id": "arxiv:2607.02057", + "pdf_url": "https://arxiv.org/pdf/2607.02057", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.01148", + "title": "Emergence of Preferential Attachment and Glass-Ceiling Effects in Autonomous Networks of LLMs", + "url": "https://arxiv.org/abs/2607.01148", + "published": "2026-07-01", + "updated": "2026-07-06", + "authors": [ + "Yiming Zhang", + "Vikram Krishnamurthy" + ], + "categories": [ + "cs.SI", + "eess.SY" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01148", + "source": "arxiv", + "source_id": "arxiv:2607.01148", + "pdf_url": "https://arxiv.org/pdf/2607.01148", + "primary_query": "llm-agent" + }, + { + "id": "2607.01426", + "title": "When Should Service Agents Reconsider? Difficulty-Routed Control in Customer-Service Operations", + "url": "https://arxiv.org/abs/2607.01426", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Qian Chen", + "Chengyuan Liu", + "Xin Yu" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.01426", + "source": "arxiv", + "source_id": "arxiv:2607.01426", + "pdf_url": "https://arxiv.org/pdf/2607.01426", + "primary_query": "tool-use" + }, + { + "id": "2606.31422", + "title": "Ask the World Before Acting: Environment Probing for Calibrated Agent World Models", + "url": "https://arxiv.org/abs/2606.31422", + "published": "2026-06-30", + "updated": "2026-07-05", + "authors": [ + "Xinyuan Song", + "Zekun Cai" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "reasoning", + "tool-use", + "world-model" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.31422", + "source": "arxiv", + "source_id": "arxiv:2606.31422", + "pdf_url": "https://arxiv.org/pdf/2606.31422", + "primary_query": "language-agent" + }, + { + "id": "2606.31055", + "title": "Reference-Based Prosody and Rhythm Evaluation for Spoken Dialogue Systems", + "url": "https://arxiv.org/abs/2606.31055", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Ashish Hallur", + "Thomas Thebaud", + "Georgi Tinchev", + "Venkatesh Ravichandran", + "Laureano Moro-Velazquez" + ], + "categories": [ + "cs.CL", + "cs.SD", + "eess.AS" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.31055", + "source": "arxiv", + "source_id": "arxiv:2606.31055", + "pdf_url": "https://arxiv.org/pdf/2606.31055", + "primary_query": "ai-agent" + }, + { + "id": "2606.30775", + "title": "A Single Rewrite Suffices: Empirical Lessons from Production Skill Description Optimization", + "url": "https://arxiv.org/abs/2606.30775", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yangqiaoyu Zhou", + "Mohammad Alqudah", + "Kwei-Herng Lai", + "Aaron Halfaker", + "Yingqi Xiong", + "Yaar Harari" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "rag", + "tool-use", + "workflow-agent" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.30775", + "source": "arxiv", + "source_id": "arxiv:2606.30775", + "pdf_url": "https://arxiv.org/pdf/2606.30775", + "primary_query": "ai-agent" + }, + { + "id": "2606.30441", + "title": "Translating Natural Language to Strategic Temporal Specifications via LLMs", + "url": "https://arxiv.org/abs/2606.30441", + "published": "2026-06-29", + "updated": "2026-07-02", + "authors": [ + "Marco Aruta", + "Francesco Improta", + "Vadim Malvone", + "Aniello Murano", + "Vladana Perlić" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30441", + "source": "arxiv", + "source_id": "arxiv:2606.30441", + "pdf_url": "https://arxiv.org/pdf/2606.30441", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28805", + "title": "Physics Models for Sim-to-Real Transfer in Professional-Level Robot Table Tennis", + "url": "https://arxiv.org/abs/2606.28805", + "published": "2026-06-27", + "updated": "2026-07-01", + "authors": [ + "Christian Conti", + "Bilan Yang", + "Alexander Sigrist", + "Lorenzo Miele", + "Yamen Saraiji", + "Peter Dürr", + "Naoya Takahashi" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.28805", + "source": "arxiv", + "source_id": "arxiv:2606.28805", + "pdf_url": "https://arxiv.org/pdf/2606.28805", + "primary_query": "ai-agent" + }, + { + "id": "2606.27944", + "title": "It Lied to a Doctor to Buy Poison Ingredients: Quantifying Real-World Misuse of Phone-use Agents", + "url": "https://arxiv.org/abs/2606.27944", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Yiming Sun", + "Chen Chen", + "Zifan Zhou", + "Mi Zhang" + ], + "categories": [ + "cs.MM", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-safety", + "computer-use", + "rag" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.27944", + "source": "arxiv", + "source_id": "arxiv:2606.27944", + "pdf_url": "https://arxiv.org/pdf/2606.27944", + "primary_query": "ai-agent" + }, + { + "id": "2606.28277", + "title": "Towards Automating Scientific Review with Google's Paper Assistant Tool", + "url": "https://arxiv.org/abs/2606.28277", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Rajesh Jayaram", + "Drew Tyler", + "David Woodruff", + "Corinna Cortes", + "Yossi Matias", + "Vahab Mirrokni", + "Vincent Cohen-Addad" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.28277", + "source": "arxiv", + "source_id": "arxiv:2606.28277", + "pdf_url": "https://arxiv.org/pdf/2606.28277", + "primary_query": "agentic-ai" + }, + { + "id": "2606.28125", + "title": "How Humans, Bots, and Agents Communicate About Vulnerabilities in Pull Requests", + "url": "https://arxiv.org/abs/2606.28125", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Pien Rooijendijk", + "Christoph Treude", + "Mairieli Wessel" + ], + "categories": [ + "cs.SE", + "cs.CR" + ], + "topics": [ + "agent-safety", + "coding-agent", + "planning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.28125", + "source": "arxiv", + "source_id": "arxiv:2606.28125", + "pdf_url": "https://arxiv.org/pdf/2606.28125", + "primary_query": "coding-agent" + }, + { + "id": "2606.26978", + "title": "To Run or Not to Run: Analyzing the Cost-Effectiveness of Code Execution in LLM-Based Program Repair", + "url": "https://arxiv.org/abs/2606.26978", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Zhihao Lin", + "Junhua Zhu", + "Mingyi Zhou", + "Xin Wang", + "Zhensu Sun", + "Renyu Yang", + "David Lo", + "Li Li" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.26978", + "source": "arxiv", + "source_id": "arxiv:2606.26978", + "pdf_url": "https://arxiv.org/pdf/2606.26978", + "primary_query": "coding-agent" + }, + { + "id": "2606.26028", + "title": "Can Trustless Agents Be Trusted? An Empirical Study of the ERC-8004 Decentralized AI Agent Ecosystem", + "url": "https://arxiv.org/abs/2606.26028", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Xihan Xiong", + "Zelin Li", + "Wei Wei", + "Qin Wang", + "William Knottenbelt", + "Zhipeng Wang" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.MA" + ], + "topics": [ + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.26028", + "source": "arxiv", + "source_id": "arxiv:2606.26028", + "pdf_url": "https://arxiv.org/pdf/2606.26028", + "primary_query": "ai-agent" + }, + { + "id": "2606.25342", + "title": "Lifelong In-Context Learning with Transformers Requires Parametric Forms of Attention", + "url": "https://arxiv.org/abs/2606.25342", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Luke McDermott", + "Robert W. Heath", + "Rahul Parhi" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.25342", + "source": "arxiv", + "source_id": "arxiv:2606.25342", + "pdf_url": "https://arxiv.org/pdf/2606.25342", + "primary_query": "ai-agent" + }, + { + "id": "2606.26175", + "title": "RMTL: Reinforced Micro-task Learning for Long-Horizon Manipulation with VLM Rewards", + "url": "https://arxiv.org/abs/2606.26175", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Anıl Can Ateş", + "Orhan Kahraman", + "Cihan Topal" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "computer-use", + "embodied-agent", + "planning", + "rag" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.26175", + "source": "arxiv", + "source_id": "arxiv:2606.26175", + "pdf_url": "https://arxiv.org/pdf/2606.26175", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.22798", + "title": "Does the Same Token Mean the Same State? MoE Routing as Signal for Reasoning Control", + "url": "https://arxiv.org/abs/2606.22798", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Kang Chen", + "Minshen Yu", + "Junjie Nian", + "Yaoning Wang", + "Yixin Cao", + "Yugang Jiang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "coding-agent", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.22798", + "source": "arxiv", + "source_id": "arxiv:2606.22798", + "pdf_url": "https://arxiv.org/pdf/2606.22798", + "primary_query": "coding-agent" + }, + { + "id": "2606.22425", + "title": "SVGym (SciVerseGym): An Environment for Reinforcement Learning and Bayesian Optimization in Crystal Discovery", + "url": "https://arxiv.org/abs/2606.22425", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Bin Cao" + ], + "categories": [ + "cs.AI", + "cond-mat.mtrl-sci" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agentic-ai", + "language-agent" + ], + "arxiv_id": "2606.22425", + "source": "arxiv", + "source_id": "arxiv:2606.22425", + "pdf_url": "https://arxiv.org/pdf/2606.22425", + "primary_query": "agentic-ai" + }, + { + "id": "2606.21811", + "title": "Steer, Don't Solve: Training Small Critic Models for Large Code Agents", + "url": "https://arxiv.org/abs/2606.21811", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Shubham Gandhi", + "Yiqing Xie", + "Atharva Naik", + "Ruichen Zhu", + "Carolyn Rose" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.LG" + ], + "topics": [ + "coding-agent", + "computer-use", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.21811", + "source": "arxiv", + "source_id": "arxiv:2606.21811", + "pdf_url": "https://arxiv.org/pdf/2606.21811", + "primary_query": "coding-agent" + }, + { + "id": "2606.21315", + "title": "Social World Model for Lifelong Social Intelligence", + "url": "https://arxiv.org/abs/2606.21315", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Yu Luo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.21315", + "source": "arxiv", + "source_id": "arxiv:2606.21315", + "pdf_url": "https://arxiv.org/pdf/2606.21315", + "primary_query": "language-agent" + }, + { + "id": "2606.21453", + "title": "CORTIS: Text-Only Adaptation of Spoken Language Models for Task-Oriented Voice Agents", + "url": "https://arxiv.org/abs/2606.21453", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Youngwon Choi", + "Hyeonyu Kim", + "Taeyoun Kwon", + "Donghyuk Jung", + "Myeongkyun Cho" + ], + "categories": [ + "cs.HC", + "cs.AI", + "cs.SD", + "eess.AS" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2606.21453", + "source": "arxiv", + "source_id": "arxiv:2606.21453", + "pdf_url": "https://arxiv.org/pdf/2606.21453", + "primary_query": "function-calling" + }, + { + "id": "2606.21654", + "title": "ChainWorld: Composing Long-Horizon Desktop Workloads from Atomic OSWorld Tasks", + "url": "https://arxiv.org/abs/2606.21654", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Vincent Siu", + "Manasi Sharma", + "Dawn Song", + "Daniel Yue Zhang", + "Chenguang Wang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.21654", + "source": "arxiv", + "source_id": "arxiv:2606.21654", + "pdf_url": "https://arxiv.org/pdf/2606.21654", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.20753", + "title": "Empowering Polymeric Materials Discovery by Artificial Intelligence", + "url": "https://arxiv.org/abs/2606.20753", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Chenyao Ma", + "Linda Zhang", + "Yuheng Chen", + "Wei Du", + "Shangwen Fang", + "Zihao Jiang", + "Chuanyu Liu", + "Xinyu Ma", + "Rui Su", + "Gang Wang", + "Muyao Yu", + "Dong Zhong", + "Jie Zhu", + "Weibo Gong", + "Huan Gu", + "Limin Li", + "Chen Shen", + "Rui Wu", + "Zhenghao Wu", + "Kan Xu", + "Min Zhou", + "Donglin He", + "Xiayun Huang", + "Shan Jiang", + "Pengfei Ou", + "Jiayu Peng", + "Yuwei Zhang", + "Jie Zhao", + "Di Zhang", + "Piao Ma", + "Zhenghao Li", + "Hao Li" + ], + "categories": [ + "physics.chem-ph", + "cs.AI" + ], + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.20753", + "source": "arxiv", + "source_id": "arxiv:2606.20753", + "pdf_url": "https://arxiv.org/pdf/2606.20753", + "primary_query": "ai-agent" + }, + { + "id": "2606.18733", + "title": "SWE-Future: Forecast-Conditioned Data Synthesis for Future-Oriented Software Engineering Agents", + "url": "https://arxiv.org/abs/2606.18733", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Qiao Zhao", + "JianYing Qu", + "Jun Zhang", + "Yehua Yang", + "Hanwen Du", + "Zhongkai Sun" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.18733", + "source": "arxiv", + "source_id": "arxiv:2606.18733", + "pdf_url": "https://arxiv.org/pdf/2606.18733", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18746", + "title": "What Must Generalist Agents Remember?", + "url": "https://arxiv.org/abs/2606.18746", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Khurram Yamin", + "Namrata Deka", + "Maitreyi Swaroop", + "Albert Ting", + "Jeff Schneider", + "Bryan Wilder" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "planning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.18746", + "source": "arxiv", + "source_id": "arxiv:2606.18746", + "pdf_url": "https://arxiv.org/pdf/2606.18746", + "primary_query": "agent-memory" + }, + { + "id": "2606.18996", + "title": "TRAP: Benchmark for Task-completion and Resistance to Active Privacy-extraction", + "url": "https://arxiv.org/abs/2606.18996", + "published": "2026-06-17", + "updated": "2026-06-18", + "authors": [ + "Moon Ye-Bin", + "Nam Hyeon-Woo", + "Baek Seong-Eun", + "Yejin Yeo", + "Tae-Hyun Oh" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.18996", + "source": "arxiv", + "source_id": "arxiv:2606.18996", + "pdf_url": "https://arxiv.org/pdf/2606.18996", + "primary_query": "tool-use" + }, + { + "id": "2606.18385", + "title": "CaVe-VLM-CoT: An Interpretable Vision-Language Model Framework", + "url": "https://arxiv.org/abs/2606.18385", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Sneha Rao", + "Shaina Raza", + "Dhanesh Ramachandram" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.18385", + "source": "arxiv", + "source_id": "arxiv:2606.18385", + "pdf_url": "https://arxiv.org/pdf/2606.18385", + "primary_query": "rag-agent" + }, + { + "id": "2606.18005", + "title": "LLM Consumer Behavior Theory: Foundations of a Novel Research Field", + "url": "https://arxiv.org/abs/2606.18005", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Manon Reusens", + "Sofie Goethals", + "David Martens" + ], + "categories": [ + "cs.AI", + "econ.GN" + ], + "topics": [ + "agent-safety", + "rag", + "world-model" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.18005", + "source": "arxiv", + "source_id": "arxiv:2606.18005", + "pdf_url": "https://arxiv.org/pdf/2606.18005", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.16215", + "title": "PACT: Privileged Trace Co-Training for Multi-Turn Tool-Use Agents", + "url": "https://arxiv.org/abs/2606.16215", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Zhenbang Du", + "Jun Luo", + "Zhiwei Zheng", + "Xiangchi Yuan", + "Kejing Xia", + "Dachuan Shi", + "Qirui Jin", + "Qijia He", + "Shaofeng Zou", + "Yingbin Liang", + "Wenke Lee" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.16215", + "source": "arxiv", + "source_id": "arxiv:2606.16215", + "pdf_url": "https://arxiv.org/pdf/2606.16215", + "primary_query": "tool-use" + }, + { + "id": "2606.15971", + "title": "SAG: SQL-Retrieval Augmented Generation with Query-Time Dynamic Hyperedges", + "url": "https://arxiv.org/abs/2606.15971", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Yuchao Wu", + "Junqin Li", + "XingCheng Liang", + "Yongjie Chen", + "Yinghao Liang", + "Linyuan Mo", + "Guanxian Li" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.15971", + "source": "arxiv", + "source_id": "arxiv:2606.15971", + "pdf_url": "https://arxiv.org/pdf/2606.15971", + "primary_query": "rag-agent" + }, + { + "id": "2606.13239", + "title": "ComAct: Reframing Professional Software Manipulation via COM-as-Action Paradigm", + "url": "https://arxiv.org/abs/2606.13239", + "published": "2026-06-11", + "updated": "2026-06-30", + "authors": [ + "Jiaxin Ai", + "Tao Hu", + "Xuemeng Yang", + "Shu Zou", + "Hairong Zhang", + "Daocheng Fu", + "Yu Yang", + "Hongbin Zhou", + "Nianchen Deng", + "Pinlong Cai", + "Zhongyuan Wang", + "Botian Shi", + "Kaipeng Zhang", + "Licheng Wen" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.13239", + "source": "arxiv", + "source_id": "arxiv:2606.13239", + "pdf_url": "https://arxiv.org/pdf/2606.13239", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.10478", + "title": "3D-CoS: A New 3D Reconstruction Paradigm Based on VLM Code Synthesis", + "url": "https://arxiv.org/abs/2606.10478", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yuhao Wang", + "Puyi Wang", + "Linjie Li", + "Zhengyuan Yang", + "Kevin Qinghong Lin", + "Yu Cheng" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.10478", + "source": "arxiv", + "source_id": "arxiv:2606.10478", + "pdf_url": "https://arxiv.org/pdf/2606.10478", + "primary_query": "rag-agent" + }, + { + "id": "2606.10875", + "title": "Pushing the Limits of LLM Tool Calling via Experiential Knowledge Integration and Activation", + "url": "https://arxiv.org/abs/2606.10875", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yupu Hao", + "Zhuoran Jin", + "Huanxuan Liao", + "Kang Liu", + "Jun Zhao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.10875", + "source": "arxiv", + "source_id": "arxiv:2606.10875", + "pdf_url": "https://arxiv.org/pdf/2606.10875", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.07924", + "title": "Decoupling Semantics and Logic: A Training-Free Coarse-to-Fine Pipeline for Video Retrieval-Augmented Generation", + "url": "https://arxiv.org/abs/2606.07924", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Jiaxin Dai", + "Zehang Wei", + "Jiamin Yan", + "Xiang Xiang" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MM" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.07924", + "source": "arxiv", + "source_id": "arxiv:2606.07924", + "pdf_url": "https://arxiv.org/pdf/2606.07924", + "primary_query": "rag-agent" + }, + { + "id": "2606.06566", + "title": "NTILC: Neural Tool Invocation via Learned Compression", + "url": "https://arxiv.org/abs/2606.06566", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Andrew Krikorian", + "Yayuan Li", + "Jason J. Corso" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2606.06566", + "source": "arxiv", + "source_id": "arxiv:2606.06566", + "pdf_url": "https://arxiv.org/pdf/2606.06566", + "primary_query": "function-calling" + }, + { + "id": "2606.03800", + "title": "Trading Human Curation for Synthetic Augmentation in RLVR", + "url": "https://arxiv.org/abs/2606.03800", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Akshansh", + "Leonardo Rosa Rodrigues", + "Michael Korostelev", + "Youssef Hassan", + "Mark E. Whiting" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2606.03800", + "source": "arxiv", + "source_id": "arxiv:2606.03800", + "pdf_url": "https://arxiv.org/pdf/2606.03800", + "primary_query": "function-calling" + }, + { + "id": "2606.01722", + "title": "Post-Deterministic Distributed Systems: A New Foundation for Trustworthy Autonomous Infrastructure", + "url": "https://arxiv.org/abs/2606.01722", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Jun He", + "Deying Yu" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.DC" + ], + "topics": [ + "computer-use", + "memory", + "planning", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.01722", + "source": "arxiv", + "source_id": "arxiv:2606.01722", + "pdf_url": "https://arxiv.org/pdf/2606.01722", + "primary_query": "agent-memory" + }, + { + "id": "2606.02245", + "title": "When Knowledge Is Not Free: Cost-Aware Evidence Selection in Retrieval-Augmented Generation", + "url": "https://arxiv.org/abs/2606.02245", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Mingyan Wu", + "Han Yang", + "Omer Ben-Porat", + "Yftah Ziser" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.02245", + "source": "arxiv", + "source_id": "arxiv:2606.02245", + "pdf_url": "https://arxiv.org/pdf/2606.02245", + "primary_query": "rag-agent" + }, + { + "id": "2606.00734", + "title": "EMA: Approximate Nearest Neighbor Search with General Attribute Filtering and Dynamic Updates", + "url": "https://arxiv.org/abs/2606.00734", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Mocheng Li", + "Baotong Lu", + "James Cheng", + "Chenhao Ma" + ], + "categories": [ + "cs.DB" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2606.00734", + "source": "arxiv", + "source_id": "arxiv:2606.00734", + "pdf_url": "https://arxiv.org/pdf/2606.00734", + "primary_query": "agent-memory" + }, + { + "id": "2605.31064", + "title": "Fighting Numerical Hallucinations via Data-centric Compilation for Online Financial QA", + "url": "https://arxiv.org/abs/2605.31064", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Hao Chen", + "Xing Tang", + "Qirui Liu", + "Weijie Shi", + "Shiwei Li", + "Fuyuan Lyu", + "Weihong Luo", + "Xiku Du", + "Xiuqiang He" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.31064", + "source": "arxiv", + "source_id": "arxiv:2605.31064", + "pdf_url": "https://arxiv.org/pdf/2605.31064", + "primary_query": "rag-agent" + }, + { + "id": "2605.30628", + "title": "The Architecture of Errors: From Universal Impossibility to Patch-Local LLM Reliability", + "url": "https://arxiv.org/abs/2605.30628", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Mikhail L. Arbuzov", + "Lee Mosbacker", + "Sisong Bei", + "Ziwei Dong", + "Dmitri Kalaev", + "Alexey Shvets" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.30628", + "source": "arxiv", + "source_id": "arxiv:2605.30628", + "pdf_url": "https://arxiv.org/pdf/2605.30628", + "primary_query": "rag-agent" + }, + { + "id": "2605.28914", + "title": "AIRGuard: Guarding Agent Actions with Runtime Authority Control", + "url": "https://arxiv.org/abs/2605.28914", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Suliu Qin", + "Haomin Zhuang", + "Yujun Zhou", + "Yufei Han", + "Xiangliang Zhang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.28914", + "source": "arxiv", + "source_id": "arxiv:2605.28914", + "pdf_url": "https://arxiv.org/pdf/2605.28914", + "primary_query": "language-agent" + }, + { + "id": "2605.25162", + "title": "STREAM: A Data-Centric Framework for Mining High-Value Task-Oriented Dialogues from Streaming Media", + "url": "https://arxiv.org/abs/2605.25162", + "published": "2026-05-24", + "updated": "2026-05-24", + "authors": [ + "Liang Xue", + "Haoyu Liu", + "Cheng Wang", + "Pengyu Chen", + "Haozhuo Zheng", + "Yang Liu" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.25162", + "source": "arxiv", + "source_id": "arxiv:2605.25162", + "pdf_url": "https://arxiv.org/pdf/2605.25162", + "primary_query": "rag-agent" + }, + { + "id": "2605.21956", + "title": "Detecting Offensive Cyber Agents: A Detection-in-Depth Approach", + "url": "https://arxiv.org/abs/2605.21956", + "published": "2026-05-21", + "updated": "2026-05-21", + "authors": [ + "Matt Mittelsteadt", + "Jam Kraprayoon", + "Robin Staes-Polet", + "Oskar Galeev", + "Jan Wehner", + "Christopher Covino", + "Shaun Ee" + ], + "categories": [ + "cs.CY" + ], + "topics": [ + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.21956", + "source": "arxiv", + "source_id": "arxiv:2605.21956", + "pdf_url": "https://arxiv.org/pdf/2605.21956", + "primary_query": "agent-safety" + }, + { + "id": "2605.22177", + "title": "Maestro: Reinforcement Learning to Orchestrate Hierarchical Model-Skill Ensembles", + "url": "https://arxiv.org/abs/2605.22177", + "published": "2026-05-21", + "updated": "2026-05-21", + "authors": [ + "Jinyang Wu", + "Guocheng Zhai", + "Ruihan Jin", + "Yuhao Shen", + "Zhengxi Lu", + "Fan Zhang", + "Haoran Luo", + "Zheng Lian", + "Zhengqi Wen", + "Jianhua Tao" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.22177", + "source": "arxiv", + "source_id": "arxiv:2605.22177", + "pdf_url": "https://arxiv.org/pdf/2605.22177", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.18988", + "title": "Surviving the Unseen: Predictive Defense for Novel Multi-Turn Multimodal Attacks", + "url": "https://arxiv.org/abs/2605.18988", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Doohee You" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "workflow-agent" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.18988", + "source": "arxiv", + "source_id": "arxiv:2605.18988", + "pdf_url": "https://arxiv.org/pdf/2605.18988", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.12978", + "title": "Useful Memories Become Faulty When Continuously Updated by LLMs", + "url": "https://arxiv.org/abs/2605.12978", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Dylan Zhang", + "Yanshan Lin", + "Zhengkun Wu", + "Yihang Sun", + "Bingxuan Li", + "Dianqi Li", + "Hao Peng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.12978", + "source": "arxiv", + "source_id": "arxiv:2605.12978", + "pdf_url": "https://arxiv.org/pdf/2605.12978", + "primary_query": "agent-memory" + }, + { + "id": "2605.13918", + "title": "CA2: Code-Aware Agent for Automated Game Testing", + "url": "https://arxiv.org/abs/2605.13918", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Valliappan Chidambaram Adaikkappan", + "Vincent Martineau", + "Joshua Romoff", + "David Meger" + ], + "categories": [ + "cs.SE", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.13918", + "source": "arxiv", + "source_id": "arxiv:2605.13918", + "pdf_url": "https://arxiv.org/pdf/2605.13918", + "primary_query": "function-calling" + }, + { + "id": "2605.14038", + "title": "Model-Adaptive Tool Necessity Reveals the Knowing-Doing Gap in LLM Tool Use", + "url": "https://arxiv.org/abs/2605.14038", + "published": "2026-05-13", + "updated": "2026-05-17", + "authors": [ + "Yize Cheng", + "Chenrui Fan", + "Mahdi JafariRaviz", + "Keivan Rezaei", + "Soheil Feizi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14038", + "source": "arxiv", + "source_id": "arxiv:2605.14038", + "pdf_url": "https://arxiv.org/pdf/2605.14038", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.13414", + "title": "TRIAGE: Evaluating Prospective Metacognitive Control in LLMs under Resource Constraints", + "url": "https://arxiv.org/abs/2605.13414", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Zabir Al Nazi", + "Shubhashis Roy Dipta" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.13414", + "source": "arxiv", + "source_id": "arxiv:2605.13414", + "pdf_url": "https://arxiv.org/pdf/2605.13414", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.13898", + "title": "Bidirectional Empowerment of Metamorphic Testing and Large Language Models: A Systematic Survey", + "url": "https://arxiv.org/abs/2605.13898", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Zheng Zheng", + "Zenghui Zhou", + "Yinwang Xu", + "Daixu Ren", + "Tsong Yueh Chen" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.13898", + "source": "arxiv", + "source_id": "arxiv:2605.13898", + "pdf_url": "https://arxiv.org/pdf/2605.13898", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.06524", + "title": "Process Matters more than Output for Distinguishing Humans from Machines", + "url": "https://arxiv.org/abs/2605.06524", + "published": "2026-05-07", + "updated": "2026-05-09", + "authors": [ + "Milena Rmus", + "Mathew D. Hardy", + "Thomas L. Griffiths", + "Mayank Agrawal" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.06524", + "source": "arxiv", + "source_id": "arxiv:2605.06524", + "pdf_url": "https://arxiv.org/pdf/2605.06524", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2603.09036", + "title": "SCALAR: Learning and Composing Skills through LLM Guided Symbolic Planning and Deep RL Grounding", + "url": "https://arxiv.org/abs/2603.09036", + "published": "2026-03-10", + "updated": "2026-03-10", + "authors": [ + "Renos Zabounidis", + "Yue Wu", + "Simon Stepputtis", + "Woojun Kim", + "Yuanzhi Li", + "Tom Mitchell", + "Katia Sycara" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.09036", + "source": "arxiv", + "source_id": "arxiv:2603.09036", + "pdf_url": "https://arxiv.org/pdf/2603.09036", + "primary_query": "language-agent" + }, + { + "id": "2603.16901", + "title": "From Language to Action in Arabic: Reliable Structured Tool Calling via Data-Centric Fine-Tuning", + "url": "https://arxiv.org/abs/2603.16901", + "published": "2026-03-04", + "updated": "2026-03-04", + "authors": [ + "Omer Nacar", + "Deema Alquffari", + "Saleh Alsharideh", + "Adeem AlOtaibi", + "Abdulaziz Alabdulkarim", + "Leen Alhazmi", + "Nada Alomar", + "Wareef Alzubaidi", + "Nada Alsultan", + "Ahmed Alrabghi", + "Demah Alhoshan", + "Rana Alsayyari", + "Hamed Alruwaili", + "Albaraa Jaafar", + "Khaled Alusmani", + "Abdulaziz Alsohimy", + "Munirah Alsubaie", + "Shahd Aldukhayil", + "Arwa Alali", + "Yazeed BinShihah", + "Razan Alsulaymi", + "Nourah Alhumaid", + "Razan Abdulsalam", + "Reem Alamoudi", + "Mohammed Alkhalifa" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.16901", + "source": "arxiv", + "source_id": "arxiv:2603.16901", + "pdf_url": "https://arxiv.org/pdf/2603.16901", + "primary_query": "function-calling" + }, + { + "id": "2602.09372", + "title": "AgentSkiller: Scaling Generalist Agent Intelligence through Semantically Integrated Cross-Domain Data Synthesis", + "url": "https://arxiv.org/abs/2602.09372", + "published": "2026-02-10", + "updated": "2026-02-10", + "authors": [ + "Zexu Sun", + "Bokai Ji", + "Hengyi Cai", + "Shuaiqiang Wang", + "Lei Wang", + "Guangxia Li", + "Xu Chen" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "planning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2602.09372", + "source": "arxiv", + "source_id": "arxiv:2602.09372", + "pdf_url": "https://arxiv.org/pdf/2602.09372", + "primary_query": "function-calling" + }, + { + "id": "2601.09292", + "title": "Blue Teaming Function-Calling Agents", + "url": "https://arxiv.org/abs/2601.09292", + "published": "2026-01-14", + "updated": "2026-01-14", + "authors": [ + "Greta Dolcetti", + "Giulio Zizzo", + "Sergio Maffeis" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.09292", + "source": "arxiv", + "source_id": "arxiv:2601.09292", + "pdf_url": "https://arxiv.org/pdf/2601.09292", + "primary_query": "function-calling" + }, + { + "id": "2601.05366", + "title": "Lost in Execution: On the Multilingual Robustness of Tool Calling in Large Language Models", + "url": "https://arxiv.org/abs/2601.05366", + "published": "2026-01-08", + "updated": "2026-06-28", + "authors": [ + "Zheng Luo", + "T Pranav Kutralingam", + "Ogochukwu N Okoani", + "Wanpeng Xu", + "Hua Wei", + "Xiyang Hu" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.05366", + "source": "arxiv", + "source_id": "arxiv:2601.05366", + "pdf_url": "https://arxiv.org/pdf/2601.05366", + "primary_query": "function-calling" + }, + { + "id": "2509.18076", + "title": "Improving Large Language Models Function Calling and Interpretability via Guided-Structured Templates", + "url": "https://arxiv.org/abs/2509.18076", + "published": "2025-09-22", + "updated": "2025-09-22", + "authors": [ + "Hy Dang", + "Tianyi Liu", + "Zhuofeng Wu", + "Jingfeng Yang", + "Haoming Jiang", + "Tao Yang", + "Pei Chen", + "Zhengyang Wang", + "Helen Wang", + "Huasheng Li", + "Bing Yin", + "Meng Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.18076", + "source": "arxiv", + "source_id": "arxiv:2509.18076", + "pdf_url": "https://arxiv.org/pdf/2509.18076", + "primary_query": "function-calling" + }, + { + "id": "2508.09125", + "title": "Complex Logical Instruction Generation", + "url": "https://arxiv.org/abs/2508.09125", + "published": "2025-08-12", + "updated": "2026-01-27", + "authors": [ + "Mian Zhang", + "Shujian Liu", + "Sixun Dong", + "Ming Yin", + "Yebowen Hu", + "Xun Wang", + "Steven Ma", + "Song Wang", + "Sathish Reddy Indurthi", + "Haoyun Deng", + "Zhiyu Zoey Chen", + "Kaiqiang Song" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 9, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.09125", + "source": "arxiv", + "source_id": "arxiv:2508.09125", + "pdf_url": "https://arxiv.org/pdf/2508.09125", + "primary_query": "function-calling" + }, + { + "id": "2607.05804", + "title": "TurnOPD: Making On-Policy Distillation Turn-Aware for Efficient Long-Horizon Agent Training", + "url": "https://arxiv.org/abs/2607.05804", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Yuhang Zhou", + "Kai Zheng", + "Haoling Li", + "Dengyun Peng", + "Can Xu", + "Jingjing Chen" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "planning" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2607.05804", + "source": "arxiv", + "source_id": "arxiv:2607.05804", + "pdf_url": "https://arxiv.org/pdf/2607.05804", + "primary_query": "language-agent" + }, + { + "id": "2607.04728", + "title": "Turning Off-Policy Tokens On-Policy: A Plug-in Approach for Improving LLM Alignment", + "url": "https://arxiv.org/abs/2607.04728", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Yu Li", + "Xiuyu Li", + "Mingyang Yi", + "Jiaxing Wang", + "zhangliangxu", + "Zhaolong Xing", + "Zhen Chen" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2607.04728", + "source": "arxiv", + "source_id": "arxiv:2607.04728", + "pdf_url": "https://arxiv.org/pdf/2607.04728", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.04542", + "title": "Auto: The AGI Compiler", + "url": "https://arxiv.org/abs/2607.04542", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Jaber Jaber", + "Osama Jaber" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04542", + "source": "arxiv", + "source_id": "arxiv:2607.04542", + "pdf_url": "https://arxiv.org/pdf/2607.04542", + "primary_query": "llm-agent" + }, + { + "id": "2607.03193", + "title": "Self-Specializing Vision-Language Transmon Chip Calibration in a Physics-Grounded Environment", + "url": "https://arxiv.org/abs/2607.03193", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Animesh Tripathy", + "Aswanth Krishnan" + ], + "categories": [ + "quant-ph", + "cs.AI", + "cs.LG" + ], + "topics": [ + "planning", + "rag", + "tool-use", + "world-model" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2607.03193", + "source": "arxiv", + "source_id": "arxiv:2607.03193", + "pdf_url": "https://arxiv.org/pdf/2607.03193", + "primary_query": "language-agent" + }, + { + "id": "2607.03451", + "title": "SkillOpt-Lite: Better and Faster Agent Self-evolution via One Line of Vibe", + "url": "https://arxiv.org/abs/2607.03451", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Yifei Shen", + "Bo Li", + "Xinjie Zhang" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.LG" + ], + "topics": [ + "coding-agent", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03451", + "source": "arxiv", + "source_id": "arxiv:2607.03451", + "pdf_url": "https://arxiv.org/pdf/2607.03451", + "primary_query": "coding-agent" + }, + { + "id": "2607.02931", + "title": "VERITAS: Towards a General-Purpose Replication Tool for Scientific Research", + "url": "https://arxiv.org/abs/2607.02931", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Haokun Liu", + "Filbert Aurelian Tjiaranata", + "Chenhao Tan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02931", + "source": "arxiv", + "source_id": "arxiv:2607.02931", + "pdf_url": "https://arxiv.org/pdf/2607.02931", + "primary_query": "coding-agent" + }, + { + "id": "2607.02217", + "title": "Affinage: genome-scale mechanistic gene annotation from the published literature", + "url": "https://arxiv.org/abs/2607.02217", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Matteo Di Bernardo", + "Iain M. Cheeseman" + ], + "categories": [ + "q-bio.GN" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "reasoning" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02217", + "source": "arxiv", + "source_id": "arxiv:2607.02217", + "pdf_url": "https://arxiv.org/pdf/2607.02217", + "primary_query": "agentic-ai" + }, + { + "id": "2607.01639", + "title": "Autonomous discovery of traffic laws with AI traffic scientists", + "url": "https://arxiv.org/abs/2607.01639", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Xingyuan Dai", + "Yue Liu", + "Xiaoyan Gong", + "Qinghai Miao", + "Junyou Shang", + "Yutong Wang", + "Chao Guo", + "Yonglin Tian", + "Yizhang Chai", + "Chao Xiang", + "Yisheng Lv", + "Fei-Yue Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "planning", + "workflow-agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.01639", + "source": "arxiv", + "source_id": "arxiv:2607.01639", + "pdf_url": "https://arxiv.org/pdf/2607.01639", + "primary_query": "agentic-ai" + }, + { + "id": "2607.00910", + "title": "Calibrating the Instrument: Controllability of an LLM-Driven Synthetic Population", + "url": "https://arxiv.org/abs/2607.00910", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Mirko Degli Esposti" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "world-model" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00910", + "source": "arxiv", + "source_id": "arxiv:2607.00910", + "pdf_url": "https://arxiv.org/pdf/2607.00910", + "primary_query": "llm-agent" + }, + { + "id": "2607.00941", + "title": "From Runtime Records to Legal Findings: An Evidentiary-Adequacy Criterion for Agentic AI Oversight", + "url": "https://arxiv.org/abs/2607.00941", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Jeroen Janssen" + ], + "categories": [ + "cs.CY" + ], + "topics": [ + "agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.00941", + "source": "arxiv", + "source_id": "arxiv:2607.00941", + "pdf_url": "https://arxiv.org/pdf/2607.00941", + "primary_query": "agentic-ai" + }, + { + "id": "2607.00316", + "title": "Evolving Intelligent Complex Systems via Intellicise Networks: Architecture, Technologies, and Pathways", + "url": "https://arxiv.org/abs/2607.00316", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Ping Zhang", + "Rui Meng", + "Xiaodong Xu", + "Song Gao", + "Zixuan Huang", + "Yaheng Wang", + "Yinqiu Liu", + "Ruichen Zhang", + "Yiming Liu", + "Kaiwen Yu", + "Yaping Sun", + "Han Meng", + "Haonan Tong", + "Huishi Song", + "Qianqian Yang", + "Shuoyao Wang", + "Lexi Xu", + "Qinghe Du", + "Geng Sun", + "Jiawen Kang", + "Gang Wu", + "Yiqing Zhou", + "Haixia Zhang", + "Zesong Fei", + "Aimin Hao", + "Ming Li" + ], + "categories": [ + "eess.SP" + ], + "topics": [ + "agent-safety", + "embodied-agent", + "planning", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.00316", + "source": "arxiv", + "source_id": "arxiv:2607.00316", + "pdf_url": "https://arxiv.org/pdf/2607.00316", + "primary_query": "agentic-ai" + }, + { + "id": "2607.01299", + "title": "HYPIC: Accelerating Hybrid-Attention LLM Serving with Position-Independent Caching", + "url": "https://arxiv.org/abs/2607.01299", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Yifei Liu", + "Juntong Wu", + "Yang Liu", + "Junhao Hu", + "Minghao Li", + "Xiaoxu Chen", + "Weihang Chen" + ], + "categories": [ + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.01299", + "source": "arxiv", + "source_id": "arxiv:2607.01299", + "pdf_url": "https://arxiv.org/pdf/2607.01299", + "primary_query": "rag-agent" + }, + { + "id": "2606.31492", + "title": "Higher-order hopping-parameter expansion by human-AI collaboration", + "url": "https://arxiv.org/abs/2606.31492", + "published": "2026-06-30", + "updated": "2026-07-06", + "authors": [ + "Masakiyo Kitazawa", + "Tatsuya Wada" + ], + "categories": [ + "hep-lat", + "hep-ph" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.31492", + "source": "arxiv", + "source_id": "arxiv:2606.31492", + "pdf_url": "https://arxiv.org/pdf/2606.31492", + "primary_query": "coding-agent" + }, + { + "id": "2606.30774", + "title": "What Drives Interactive Improvement from Feedback?", + "url": "https://arxiv.org/abs/2606.30774", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Bartłomiej Cupiał", + "Jan Łojek", + "Mikołaj Garstecki", + "Szymon Pobłocki", + "Alicja Ziarko", + "Piotr Miłoś" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.30774", + "source": "arxiv", + "source_id": "arxiv:2606.30774", + "pdf_url": "https://arxiv.org/pdf/2606.30774", + "primary_query": "language-agent" + }, + { + "id": "2606.29916", + "title": "EVAF: A Test-Retest Protocol for Selective Parametric Consolidation", + "url": "https://arxiv.org/abs/2606.29916", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Haoliang Han" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.29916", + "source": "arxiv", + "source_id": "arxiv:2606.29916", + "pdf_url": "https://arxiv.org/pdf/2606.29916", + "primary_query": "language-agent" + }, + { + "id": "2606.30963", + "title": "Loc2Repair: A Framework for Evaluating the Impact of File-Level Issue Localization in Repo-Level LLM Repair", + "url": "https://arxiv.org/abs/2606.30963", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Mohammad Nour Al Awad", + "Sergey Ivanov" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.30963", + "source": "arxiv", + "source_id": "arxiv:2606.30963", + "pdf_url": "https://arxiv.org/pdf/2606.30963", + "primary_query": "coding-agent" + }, + { + "id": "2606.29556", + "title": "Persona-Trained Monte Carlo: Estimating Market-Outcome Distributions via Swarms of Persona-Conditioned Neural Policy Bots in a Limit Order Book", + "url": "https://arxiv.org/abs/2606.29556", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Salavat Ishbulatov" + ], + "categories": [ + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.29556", + "source": "arxiv", + "source_id": "arxiv:2606.29556", + "pdf_url": "https://arxiv.org/pdf/2606.29556", + "primary_query": "coding-agent" + }, + { + "id": "2606.28690", + "title": "Formal Security Analysis of Agent Protocol Composition", + "url": "https://arxiv.org/abs/2606.28690", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Shenghan Zheng", + "Qifan Zhang", + "Zheng Zhang", + "Haonan Li", + "Christophe Hauser" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.28690", + "source": "arxiv", + "source_id": "arxiv:2606.28690", + "pdf_url": "https://arxiv.org/pdf/2606.28690", + "primary_query": "ai-agent" + }, + { + "id": "2606.30678", + "title": "NanoVer: An open-source framework for interactive molecular dynamics in extended reality (iMD-XR) on commodity hardware", + "url": "https://arxiv.org/abs/2606.30678", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Mark D. Wonnacott", + "Luis Ernesto Toledo Castro", + "Harry J. Stroud", + "Ludovica Aisa", + "Mohamed Dhouioui", + "Rhoslyn Roebuck Williams", + "Denis Protopopov", + "Sila Sobrado", + "David R. Glowacki" + ], + "categories": [ + "physics.chem-ph", + "physics.bio-ph", + "physics.ed-ph" + ], + "topics": [ + "computer-use", + "reasoning", + "tool-use", + "world-model" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.30678", + "source": "arxiv", + "source_id": "arxiv:2606.30678", + "pdf_url": "https://arxiv.org/pdf/2606.30678", + "primary_query": "ai-agent" + }, + { + "id": "2606.25550", + "title": "On the Viability of Requirements Generation From Code: An Experience Report", + "url": "https://arxiv.org/abs/2606.25550", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Alexander Korn", + "Jone Bartel", + "Max Unterbusch", + "Andreas Vogelsang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.25550", + "source": "arxiv", + "source_id": "arxiv:2606.25550", + "pdf_url": "https://arxiv.org/pdf/2606.25550", + "primary_query": "rag-agent" + }, + { + "id": "2606.23679", + "title": "Semantic Browsing: Controllable Diversity for Image Generation", + "url": "https://arxiv.org/abs/2606.23679", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Sara Dorfman", + "Maya Vishnevsky", + "Omer Dahary", + "Or Patashnik", + "Daniel Cohen-Or" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.GR", + "cs.LG" + ], + "topics": [ + "rag", + "workflow-agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.23679", + "source": "arxiv", + "source_id": "arxiv:2606.23679", + "pdf_url": "https://arxiv.org/pdf/2606.23679", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20161", + "title": "ARTEMIS: Agent-guided Reliability-aware Temporal Mask Evolution for Imperfectly Supervised Video Polyp Segmentation", + "url": "https://arxiv.org/abs/2606.20161", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Tong Wang", + "Siwen Wang", + "Yaolei Qi", + "Jinxing Zhou", + "Yuting He", + "Guanyu Yang", + "Yutong Xie" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "computer-use", + "multi-agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.20161", + "source": "arxiv", + "source_id": "arxiv:2606.20161", + "pdf_url": "https://arxiv.org/pdf/2606.20161", + "primary_query": "language-agent" + }, + { + "id": "2606.20453", + "title": "Directors Duties in the Age of Agentic Artificial Intelligence", + "url": "https://arxiv.org/abs/2606.20453", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Deirdre Ahern" + ], + "categories": [ + "cs.CY", + "cs.HC" + ], + "topics": [ + "agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.20453", + "source": "arxiv", + "source_id": "arxiv:2606.20453", + "pdf_url": "https://arxiv.org/pdf/2606.20453", + "primary_query": "agentic-ai" + }, + { + "id": "2606.19670", + "title": "PiMiX 2.0: AI-enhanced Data Fusion for Radiographic Imaging and Tomography", + "url": "https://arxiv.org/abs/2606.19670", + "published": "2026-06-18", + "updated": "2026-06-20", + "authors": [ + "Zhehui Wang", + "Shanny Lin", + "Nicholas Amano", + "Susan S. Glenn", + "Ramya Gurunathan", + "Katie Liu", + "Nathan E. Peterson", + "Michelle A. Espy", + "Adam Thompson", + "Amy J. Clarke", + "Ray T. Chen" + ], + "categories": [ + "physics.ins-det", + "physics.data-an" + ], + "topics": [ + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.19670", + "source": "arxiv", + "source_id": "arxiv:2606.19670", + "pdf_url": "https://arxiv.org/pdf/2606.19670", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20708", + "title": "Simulated Customers Never Walk Away: Decision Fidelity of LLM User Simulators Measured Against Real Purchase Outcomes", + "url": "https://arxiv.org/abs/2606.20708", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Liang Chen" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "world-model" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.20708", + "source": "arxiv", + "source_id": "arxiv:2606.20708", + "pdf_url": "https://arxiv.org/pdf/2606.20708", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.16903", + "title": "Directory-Aware Query and Maintenance in Vector Databases", + "url": "https://arxiv.org/abs/2606.16903", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Mengzhao Wang", + "Zheng Gong", + "Jingpei Hu", + "Jiajie Fu", + "Maojia Sheng", + "Junwen Chen", + "Yifan Zhu" + ], + "categories": [ + "cs.DB" + ], + "topics": [ + "agent-evaluation", + "rag", + "workflow-agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.16903", + "source": "arxiv", + "source_id": "arxiv:2606.16903", + "pdf_url": "https://arxiv.org/pdf/2606.16903", + "primary_query": "agent-memory" + }, + { + "id": "2606.12485", + "title": "Speculative Rollback Correction for Quality-Diverse Web Agent Imitation", + "url": "https://arxiv.org/abs/2606.12485", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Longkun Hao", + "Hongyu Lin", + "Hao Li", + "Zhichao Yang", + "Haojie Hao", + "Dongshuo Huang", + "Haitao Yang", + "Hongyu Ge", + "Ming jie Xie", + "Yanjun Wu", + "Zi Hao Yin", + "Yan Bai", + "Yihang Lou" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "computer-use", + "reasoning", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.12485", + "source": "arxiv", + "source_id": "arxiv:2606.12485", + "pdf_url": "https://arxiv.org/pdf/2606.12485", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.08944", + "title": "LongRTL: Graph-Similarity-Guided LLM-driven Long Context RTL Optimization", + "url": "https://arxiv.org/abs/2606.08944", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Yuyang Ye", + "Che-Kuan Shen", + "Xiangfei Hu", + "Yuchen Liu", + "Shuo Yin", + "Xufeng Yao", + "Bei Yu", + "Tsung-Yi Ho" + ], + "categories": [ + "cs.AR", + "cs.PL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.08944", + "source": "arxiv", + "source_id": "arxiv:2606.08944", + "pdf_url": "https://arxiv.org/pdf/2606.08944", + "primary_query": "rag-agent" + }, + { + "id": "2606.08755", + "title": "Co-Evolving Skill Generation and Policy Optimization", + "url": "https://arxiv.org/abs/2606.08755", + "published": "2026-06-07", + "updated": "2026-06-07", + "authors": [ + "Zhiwei Zhang", + "Yudi Lin", + "Nikki Lijing Kuang", + "Linlin Wu", + "Xiaomin Li", + "Songtao Liu", + "Fenglong Ma" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.08755", + "source": "arxiv", + "source_id": "arxiv:2606.08755", + "pdf_url": "https://arxiv.org/pdf/2606.08755", + "primary_query": "language-agent" + }, + { + "id": "2606.04751", + "title": "FALSIFYBENCH: Evaluating Inductive Reasoning in LLMs with Rule Discovery Games", + "url": "https://arxiv.org/abs/2606.04751", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Leonardo Bertolazzi", + "Katya Tentori", + "Raffaella Bernardi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.04751", + "source": "arxiv", + "source_id": "arxiv:2606.04751", + "pdf_url": "https://arxiv.org/pdf/2606.04751", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.03777", + "title": "From Control Boundary to Insurance Claim: Reconstructing AI-Mediated Losses Through the CER Framework", + "url": "https://arxiv.org/abs/2606.03777", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Alex Leung", + "Rex Zhang", + "Kentaroh Toyoda", + "SiewMei Loh" + ], + "categories": [ + "cs.AI", + "cs.CR", + "q-fin.RM" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.03777", + "source": "arxiv", + "source_id": "arxiv:2606.03777", + "pdf_url": "https://arxiv.org/pdf/2606.03777", + "primary_query": "rag-agent" + }, + { + "id": "2606.02483", + "title": "Ghost Tool Calls: Issue-Time Privacy for Speculative Agent Tools", + "url": "https://arxiv.org/abs/2606.02483", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Bardia Mohammadi", + "Lars Klein", + "Akhil Arora", + "Laurent Bindschaedler" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.02483", + "source": "arxiv", + "source_id": "arxiv:2606.02483", + "pdf_url": "https://arxiv.org/pdf/2606.02483", + "primary_query": "language-agent" + }, + { + "id": "2606.01212", + "title": "DiscourseFlip: An Oblique Discourse-Level Opinion Manipulation Attack against Black-box Retrieval-Augmented Generation", + "url": "https://arxiv.org/abs/2606.01212", + "published": "2026-05-31", + "updated": "2026-06-03", + "authors": [ + "Yuyang Gong", + "Miaokun Chen", + "Jiawei Liu", + "Zhuo Chen", + "Guoxiu He", + "Wei Lu", + "XiaoFeng Wang", + "Xiaozhong Liu" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.CR", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.01212", + "source": "arxiv", + "source_id": "arxiv:2606.01212", + "pdf_url": "https://arxiv.org/pdf/2606.01212", + "primary_query": "rag-agent" + }, + { + "id": "2605.26754", + "title": "Cordon-MAS: Defending RAG against Knowledge Poisoning via Information-Flow Control", + "url": "https://arxiv.org/abs/2605.26754", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Zhe Yu", + "Wenpeng Xing", + "Gaolei Li", + "Shuguang Xiong", + "Hongzhi Wang", + "Xuyang Teng", + "Meng Han" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.26754", + "source": "arxiv", + "source_id": "arxiv:2605.26754", + "pdf_url": "https://arxiv.org/pdf/2605.26754", + "primary_query": "rag-agent" + }, + { + "id": "2605.25379", + "title": "EfficientGraph-RAG: Structured Retrieval-State Management for Cross-Task Retrieval-Augmented Generation", + "url": "https://arxiv.org/abs/2605.25379", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Miaohe Niu", + "Lianlei Shan", + "Zhengtao Yu", + "Jingbo Zhu", + "Tong Xiao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.25379", + "source": "arxiv", + "source_id": "arxiv:2605.25379", + "pdf_url": "https://arxiv.org/pdf/2605.25379", + "primary_query": "rag-agent" + }, + { + "id": "2605.21401", + "title": "Open-source LLMs administer maximum electric shocks in a Milgram-like obedience experiment", + "url": "https://arxiv.org/abs/2605.21401", + "published": "2026-05-20", + "updated": "2026-06-23", + "authors": [ + "Roland Pihlakas", + "Jan Llenzl Dagohoy" + ], + "categories": [ + "cs.CY", + "cs.AI" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.21401", + "source": "arxiv", + "source_id": "arxiv:2605.21401", + "pdf_url": "https://arxiv.org/pdf/2605.21401", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.18414", + "title": "Prompts Don't Protect: Architectural Enforcement via MCP Proxy for LLM Tool Access Control", + "url": "https://arxiv.org/abs/2605.18414", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Rohith Uppala" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.18414", + "source": "arxiv", + "source_id": "arxiv:2605.18414", + "pdf_url": "https://arxiv.org/pdf/2605.18414", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.14588", + "title": "Silent Collapse in Recursive Learning Systems", + "url": "https://arxiv.org/abs/2605.14588", + "published": "2026-05-14", + "updated": "2026-05-19", + "authors": [ + "Zhipeng Zhang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14588", + "source": "arxiv", + "source_id": "arxiv:2605.14588", + "pdf_url": "https://arxiv.org/pdf/2605.14588", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.00651", + "title": "EQSANS-CLI: A natural-language, agent-ready command-line tool for small-angle neutron scattering data reduction at EQ-SANS", + "url": "https://arxiv.org/abs/2605.00651", + "published": "2026-05-01", + "updated": "2026-05-01", + "authors": [ + "Changwoo Do" + ], + "categories": [ + "physics.ins-det" + ], + "topics": [ + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.00651", + "source": "arxiv", + "source_id": "arxiv:2605.00651", + "pdf_url": "https://arxiv.org/pdf/2605.00651", + "primary_query": "language-agent" + }, + { + "id": "2605.01047", + "title": "LLM Ghostbusters: Surgical Hallucination Suppression via Adaptive Unlearning", + "url": "https://arxiv.org/abs/2605.01047", + "published": "2026-05-01", + "updated": "2026-05-01", + "authors": [ + "Joseph Spracklen", + "Pedram Aghazadeh", + "Farinaz Koushanfar", + "Murtuza Jadliwala" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.01047", + "source": "arxiv", + "source_id": "arxiv:2605.01047", + "pdf_url": "https://arxiv.org/pdf/2605.01047", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.28138", + "title": "Crab: A Semantics-Aware Checkpoint/Restore Runtime for Agent Sandboxes", + "url": "https://arxiv.org/abs/2604.28138", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Tianyuan Wu", + "Chaokun Chang", + "Lunxi Cao", + "Wei Gao", + "Wei Wang" + ], + "categories": [ + "cs.OS", + "cs.AI" + ], + "topics": [ + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.28138", + "source": "arxiv", + "source_id": "arxiv:2604.28138", + "pdf_url": "https://arxiv.org/pdf/2604.28138", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2603.10285", + "title": "Conversational AI-Enhanced Exploration System to Query Large-Scale Digitised Collections of Natural History Museums", + "url": "https://arxiv.org/abs/2603.10285", + "published": "2026-03-11", + "updated": "2026-03-11", + "authors": [ + "Yiyuan Wang", + "Andrew Johnston", + "Zoë Sadokierski", + "Rhiannon Stephens", + "Shane T. Ahyong" + ], + "categories": [ + "cs.HC", + "cs.AI", + "cs.CY", + "cs.DL", + "cs.ET" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.10285", + "source": "arxiv", + "source_id": "arxiv:2603.10285", + "pdf_url": "https://arxiv.org/pdf/2603.10285", + "primary_query": "function-calling" + }, + { + "id": "2602.08121", + "title": "Initial Risk Probing and Feasibility Testing of Glow: a Generative AI-Powered Dialectical Behavior Therapy Skills Coach for Substance Use Recovery and HIV Prevention", + "url": "https://arxiv.org/abs/2602.08121", + "published": "2026-02-08", + "updated": "2026-02-08", + "authors": [ + "Liying Wang", + "Madison Lee", + "Yunzhang Jiang", + "Steven Chen", + "Kewei Sha", + "Yunhe Feng", + "Frank Wong", + "Lisa Hightow-Weidman", + "Weichao Yuwen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.08121", + "source": "arxiv", + "source_id": "arxiv:2602.08121", + "pdf_url": "https://arxiv.org/pdf/2602.08121", + "primary_query": "agent-safety" + }, + { + "id": "2601.06937", + "title": "mind_call: A Dataset for Mental Health Function Calling with Large Language Models", + "url": "https://arxiv.org/abs/2601.06937", + "published": "2026-01-11", + "updated": "2026-01-11", + "authors": [ + "Fozle Rabbi Shafi", + "M. Anwar Hossain", + "Salimur Choudhury" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "rag", + "reasoning", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.06937", + "source": "arxiv", + "source_id": "arxiv:2601.06937", + "pdf_url": "https://arxiv.org/pdf/2601.06937", + "primary_query": "function-calling" + }, + { + "id": "2510.13558", + "title": "Steer-MoE: Efficient Audio-Language Alignment with a Mixture-of-Experts Steering Module", + "url": "https://arxiv.org/abs/2510.13558", + "published": "2025-10-15", + "updated": "2025-10-15", + "authors": [ + "Ruitao Feng", + "Bixi Zhang", + "Sheng Liang", + "Zheng Yuan" + ], + "categories": [ + "cs.SD" + ], + "topics": [ + "agent-safety", + "reasoning" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.13558", + "source": "arxiv", + "source_id": "arxiv:2510.13558", + "pdf_url": "https://arxiv.org/pdf/2510.13558", + "primary_query": "function-calling" + }, + { + "id": "2509.26463", + "title": "ErrorPrism: Reconstructing Error Propagation Paths in Cloud Service Systems", + "url": "https://arxiv.org/abs/2509.26463", + "published": "2025-09-30", + "updated": "2025-09-30", + "authors": [ + "Junsong Pu", + "Yichen Li", + "Zhuangbin Chen", + "Jinyang Liu", + "Zhihan Jiang", + "Jianjun Chen", + "Rui Shi", + "Zibin Zheng", + "Tieying Zhang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.26463", + "source": "arxiv", + "source_id": "arxiv:2509.26463", + "pdf_url": "https://arxiv.org/pdf/2509.26463", + "primary_query": "function-calling" + }, + { + "id": "2509.24229", + "title": "Model Fusion with Multi-LoRA Inference for Tool-Enhanced Game Dialogue Agents", + "url": "https://arxiv.org/abs/2509.24229", + "published": "2025-09-29", + "updated": "2025-09-29", + "authors": [ + "Kangxu Wang", + "Ze Chen", + "Chengcheng Wei", + "Jiewen Zheng", + "Jiarong He", + "Max Gao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.24229", + "source": "arxiv", + "source_id": "arxiv:2509.24229", + "pdf_url": "https://arxiv.org/pdf/2509.24229", + "primary_query": "function-calling" + }, + { + "id": "2509.04518", + "title": "Advancing SLM Tool-Use Capability using Reinforcement Learning", + "url": "https://arxiv.org/abs/2509.04518", + "published": "2025-09-03", + "updated": "2025-09-08", + "authors": [ + "Dhruvi Paprunia", + "Vansh Kharidia", + "Pankti Doshi" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "tool-use" + ], + "score": 8, + "relevance": "medium", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.04518", + "source": "arxiv", + "source_id": "arxiv:2509.04518", + "pdf_url": "https://arxiv.org/pdf/2509.04518", + "primary_query": "function-calling" + }, + { + "id": "2607.04103", + "title": "Governing Generative AI Across Financial Institutions: An SR 26-2-Compatible Framework for Generative AI Risk Control", + "url": "https://arxiv.org/abs/2607.04103", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Yiqing Wang", + "Yixin Kang", + "Luyun Lin", + "Siqi Mao" + ], + "categories": [ + "q-fin.RM", + "cs.LG" + ], + "topics": [ + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04103", + "source": "arxiv", + "source_id": "arxiv:2607.04103", + "pdf_url": "https://arxiv.org/pdf/2607.04103", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03181", + "title": "Teaming Up with AI: Coordination and Cooperation", + "url": "https://arxiv.org/abs/2607.03181", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Nicole Immorlica", + "Inbal Talgam-Cohen" + ], + "categories": [ + "cs.GT", + "cs.AI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.03181", + "source": "arxiv", + "source_id": "arxiv:2607.03181", + "pdf_url": "https://arxiv.org/pdf/2607.03181", + "primary_query": "ai-agent" + }, + { + "id": "2607.03100", + "title": "Flow-A11y: Flow-Aware Accessibility Testing", + "url": "https://arxiv.org/abs/2607.03100", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Nasr Eddine Fliti", + "Leisan Kokorina", + "Florian Tambon", + "Michael Papadakis" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.03100", + "source": "arxiv", + "source_id": "arxiv:2607.03100", + "pdf_url": "https://arxiv.org/pdf/2607.03100", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.00613", + "title": "Ai2-Kit: Streamlining AI-Accelerated Ab Initio Workflows for Complex Chemical Systems", + "url": "https://arxiv.org/abs/2607.00613", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Sheng Bi", + "Wei-Hong Xu", + "Yong-Bin Zhuang", + "Jia-Xin Zhu", + "Jiang-Peng Qiu", + "Yu-Hang Tang", + "Xiang-Long Du", + "Qi You", + "Yun-Pei Liu", + "Fu-Qiang Gong", + "Yu-Xin Guo", + "Yi-Ze Wang", + "Cheng-Xuan Wang", + "Zi-Heng Gong", + "Zi-Qiang Chen", + "Chang Liu", + "Si-Yuan Han", + "Jian Gu", + "Jia-Xin Li", + "Yi-Ming Chen", + "Lin Huang", + "Si-Jie Chen", + "Bo-Ying Huang", + "Jie-Zhen Xia", + "Fan-Jie Xu", + "Su-Yang Zhong", + "Peng-Wei Xu", + "Jun-Yi Wang", + "Xing-Yun Xie", + "Yu-Lei Gong", + "Yan-Yi Su", + "Yue Liu", + "Rui-Hao Bi", + "Lang Li", + "Fei-Teng Wang", + "Jing-Xiang Zou", + "Mei Jia", + "Jie-Qiong Li", + "Min Lin", + "Qi-Yuan Fan", + "Juan-Juan Sun", + "Jia-Bo Le", + "Zixuan Wei", + "Jin-Yuan Hu", + "Meng-Lei Jia", + "Yan Sun", + "Xiao-Hui Yang", + "Fujie Tang", + "Feng Wang", + "Jun Cheng" + ], + "categories": [ + "physics.chem-ph" + ], + "topics": [ + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.00613", + "source": "arxiv", + "source_id": "arxiv:2607.00613", + "pdf_url": "https://arxiv.org/pdf/2607.00613", + "primary_query": "ai-agent" + }, + { + "id": "2606.31399", + "title": "World-Model Collapse as a Phase Transition", + "url": "https://arxiv.org/abs/2606.31399", + "published": "2026-06-30", + "updated": "2026-07-04", + "authors": [ + "Xinyuan Song", + "Zekun Cai" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "planning", + "tool-use", + "world-model" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.31399", + "source": "arxiv", + "source_id": "arxiv:2606.31399", + "pdf_url": "https://arxiv.org/pdf/2606.31399", + "primary_query": "language-agent" + }, + { + "id": "2607.00155", + "title": "A Contextual-Bandit Oversight Game with Two-Sided Informational Asymmetry", + "url": "https://arxiv.org/abs/2607.00155", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yunjin Tong" + ], + "categories": [ + "cs.AI", + "cs.GT" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "tool-use" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.00155", + "source": "arxiv", + "source_id": "arxiv:2607.00155", + "pdf_url": "https://arxiv.org/pdf/2607.00155", + "primary_query": "ai-agent" + }, + { + "id": "2606.31214", + "title": "EasyScan_HEP 2: Agent-Ready Parameter Scans for High-Energy Physics", + "url": "https://arxiv.org/abs/2606.31214", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yang Xiao", + "Yuanfang Yue", + "Yang Zhang" + ], + "categories": [ + "hep-ph" + ], + "topics": [ + "workflow-agent" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.31214", + "source": "arxiv", + "source_id": "arxiv:2606.31214", + "pdf_url": "https://arxiv.org/pdf/2606.31214", + "primary_query": "ai-agent" + }, + { + "id": "2606.29389", + "title": "Exploring the Cryptographic Limits of Transformer Networks", + "url": "https://arxiv.org/abs/2606.29389", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Stefan Domunco", + "Andis Draguns", + "Philip Torr", + "Isaac Robinson", + "Christian Schroeder de Witt" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.29389", + "source": "arxiv", + "source_id": "arxiv:2606.29389", + "pdf_url": "https://arxiv.org/pdf/2606.29389", + "primary_query": "ai-agent" + }, + { + "id": "2606.29100", + "title": "Toward Exascale AI for Science: A Scalable AI Skill for Autonomous Microkinetics Discovery", + "url": "https://arxiv.org/abs/2606.29100", + "published": "2026-06-27", + "updated": "2026-07-03", + "authors": [ + "Ken-ichi Nomura", + "William Dawson", + "Nabankur Dasgupta", + "Taufeq Mohammed Razakh", + "Thomas Linker", + "Kai Ito", + "Aiichiro Nakano" + ], + "categories": [ + "cs.CE" + ], + "topics": [ + "agent-evaluation", + "workflow-agent", + "world-model" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.29100", + "source": "arxiv", + "source_id": "arxiv:2606.29100", + "pdf_url": "https://arxiv.org/pdf/2606.29100", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26448", + "title": "Closing the Loop to Discover Psychological Theories with an Automated Cognitive Scientist", + "url": "https://arxiv.org/abs/2606.26448", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Akshay K. Jagadish", + "Younes Strittmatter", + "Nori Jacoby", + "George Kachergis", + "Eric Schulz", + "Nathaniel Daw", + "Suyog H. Chandramouli", + "Thomas L. Griffiths" + ], + "categories": [ + "q-bio.NC", + "cs.AI" + ], + "topics": [ + "agent" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "agentic-ai", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.26448", + "source": "arxiv", + "source_id": "arxiv:2606.26448", + "pdf_url": "https://arxiv.org/pdf/2606.26448", + "primary_query": "agentic-ai" + }, + { + "id": "2606.25525", + "title": "The impact of artificial intelligence on enterprise software user roles", + "url": "https://arxiv.org/abs/2606.25525", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Isabel Unger", + "Elizangela Valarini", + "Martin Schrepp", + "Nina Hollender", + "Gabriela Rocha", + "Erik Bertram" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.25525", + "source": "arxiv", + "source_id": "arxiv:2606.25525", + "pdf_url": "https://arxiv.org/pdf/2606.25525", + "primary_query": "agentic-ai" + }, + { + "id": "2606.25496", + "title": "Recommendation as Generation: Unifying Personalized Video Generation and Recommendation at Industrial Scale", + "url": "https://arxiv.org/abs/2606.25496", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yanhua Cheng", + "Bo Wang", + "Haotian Zhang", + "Xinyuan Gao", + "Zhihui Yin", + "Ben Xue", + "Yongzhi Li", + "Jieting Xue", + "Ye Ma", + "Minquan Wang", + "Jiahui Li", + "Tianyu Xu", + "Zhiqiang Liu", + "Xiao Lin", + "Shiyang Wen", + "Changcheng Li", + "Liu Liu", + "Quan Chen", + "Peng Jiang", + "Kun Gai" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.25496", + "source": "arxiv", + "source_id": "arxiv:2606.25496", + "pdf_url": "https://arxiv.org/pdf/2606.25496", + "primary_query": "rag-agent" + }, + { + "id": "2606.23348", + "title": "Superhuman AI for Generals.io Using Self-Play Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.23348", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Matej Straka", + "Viliam Lisý", + "Martin Schmid" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.23348", + "source": "arxiv", + "source_id": "arxiv:2606.23348", + "pdf_url": "https://arxiv.org/pdf/2606.23348", + "primary_query": "ai-agent" + }, + { + "id": "2606.21124", + "title": "PulseCX: Breaking the Closed-World Assumption in Real-Time CX", + "url": "https://arxiv.org/abs/2606.21124", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Rajat Agarwal", + "Suvidha Tripathi", + "Shubham Sharma" + ], + "categories": [ + "cs.AI", + "cs.IR" + ], + "topics": [ + "memory", + "tool-use" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.21124", + "source": "arxiv", + "source_id": "arxiv:2606.21124", + "pdf_url": "https://arxiv.org/pdf/2606.21124", + "primary_query": "ai-agent" + }, + { + "id": "2606.16319", + "title": "Architectural Wisdom: A Framework for Governing Optimization in AI Systems", + "url": "https://arxiv.org/abs/2606.16319", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Edward Y. Chang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.16319", + "source": "arxiv", + "source_id": "arxiv:2606.16319", + "pdf_url": "https://arxiv.org/pdf/2606.16319", + "primary_query": "tool-use" + }, + { + "id": "2605.18991", + "title": "Agent Security is a Systems Problem", + "url": "https://arxiv.org/abs/2605.18991", + "published": "2026-05-18", + "updated": "2026-05-20", + "authors": [ + "Mihai Christodorescu", + "Earlence Fernandes", + "Ashish Hooda", + "Somesh Jha", + "Johann Rehberger", + "Kamalika Chaudhuri", + "Xiaohan Fu", + "Khawaja Shams", + "Guy Amir", + "Jihye Choi", + "Sarthak Choudhary", + "Nils Palumbo", + "Andrey Labunets", + "Nishit V. Pandya" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.18991", + "source": "arxiv", + "source_id": "arxiv:2605.18991", + "pdf_url": "https://arxiv.org/pdf/2605.18991", + "primary_query": "agent-safety" + }, + { + "id": "2605.02028", + "title": "Language models fail at extended rule following", + "url": "https://arxiv.org/abs/2605.02028", + "published": "2026-05-03", + "updated": "2026-05-16", + "authors": [ + "Tianxiang Dai", + "Jonathan Fan" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "tool-use" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.02028", + "source": "arxiv", + "source_id": "arxiv:2605.02028", + "pdf_url": "https://arxiv.org/pdf/2605.02028", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.25610", + "title": "Optimizing ground state preparation protocols with autoresearch", + "url": "https://arxiv.org/abs/2604.25610", + "published": "2026-04-28", + "updated": "2026-05-08", + "authors": [ + "Luis Mantilla Calderón", + "Jérôme F. Gonthier", + "Ignacio Gustin", + "Varinia Bernales", + "Alán Aspuru-Guzik" + ], + "categories": [ + "quant-ph" + ], + "topics": [ + "coding-agent", + "computer-use", + "world-model" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.25610", + "source": "arxiv", + "source_id": "arxiv:2604.25610", + "pdf_url": "https://arxiv.org/pdf/2604.25610", + "primary_query": "language-agent" + }, + { + "id": "2603.25636", + "title": "Designing Any Imaging System from Natural Language: Agent-Constrained Composition over a Finite Primitive Basis", + "url": "https://arxiv.org/abs/2603.25636", + "published": "2026-03-26", + "updated": "2026-03-26", + "authors": [ + "Chengshuai Yang" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "planning", + "tool-use" + ], + "score": 7, + "relevance": "medium", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.25636", + "source": "arxiv", + "source_id": "arxiv:2603.25636", + "pdf_url": "https://arxiv.org/pdf/2603.25636", + "primary_query": "language-agent" + }, + { + "id": "2607.05835", + "title": "Tangent classes of matroids and wonderful compactifications", + "url": "https://arxiv.org/abs/2607.05835", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Ronnie Cheng", + "Shurui Liu", + "Guoxiong Gao" + ], + "categories": [ + "math.AG", + "cs.AI", + "math.CO" + ], + "topics": [ + "computer-use", + "reasoning" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.05835", + "source": "arxiv", + "source_id": "arxiv:2607.05835", + "pdf_url": "https://arxiv.org/pdf/2607.05835", + "primary_query": "ai-agent" + }, + { + "id": "2607.05744", + "title": "Unicode TAG-Block Concealment of Tool-Metadata Payloads in the Model Context Protocol: An Approval-View Fidelity Gap Across Three Independent Server Implementations", + "url": "https://arxiv.org/abs/2607.05744", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Mohammadreza Rashidi" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.SE" + ], + "topics": [ + "coding-agent", + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.05744", + "source": "arxiv", + "source_id": "arxiv:2607.05744", + "pdf_url": "https://arxiv.org/pdf/2607.05744", + "primary_query": "coding-agent" + }, + { + "id": "2607.02905", + "title": "Pre-Strings Lectures on Artificial Intelligence", + "url": "https://arxiv.org/abs/2607.02905", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "James Halverson" + ], + "categories": [ + "hep-th" + ], + "topics": [ + "agent-evaluation", + "workflow-agent" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02905", + "source": "arxiv", + "source_id": "arxiv:2607.02905", + "pdf_url": "https://arxiv.org/pdf/2607.02905", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02609", + "title": "Knowledge-Centric Information Systems", + "url": "https://arxiv.org/abs/2607.02609", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Mariano Garralda-Barrio" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.DB" + ], + "topics": [ + "workflow-agent" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02609", + "source": "arxiv", + "source_id": "arxiv:2607.02609", + "pdf_url": "https://arxiv.org/pdf/2607.02609", + "primary_query": "agentic-ai" + }, + { + "id": "2607.01415", + "title": "The Rollout Infrastructure Tax in Coding-Agent Reinforcement Learning", + "url": "https://arxiv.org/abs/2607.01415", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Daniel Thi Graviet", + "Lovre Pesut", + "Ivan Dagelic", + "Vedran Jukic", + "Ivan Burazin" + ], + "categories": [ + "cs.LG", + "cs.DC" + ], + "topics": [ + "coding-agent" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.01415", + "source": "arxiv", + "source_id": "arxiv:2607.01415", + "pdf_url": "https://arxiv.org/pdf/2607.01415", + "primary_query": "coding-agent" + }, + { + "id": "2607.01380", + "title": "Lagrangian evaluation of polymeric stress in viscoelastic fluids", + "url": "https://arxiv.org/abs/2607.01380", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Mohammad Majidi", + "Rishu Gandhi", + "Louison Thorens", + "Maliheh Teimouri", + "Jeffrey S. Guasto", + "Arezoo M. Ardekani" + ], + "categories": [ + "physics.flu-dyn" + ], + "topics": [ + "agent-evaluation", + "world-model" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.01380", + "source": "arxiv", + "source_id": "arxiv:2607.01380", + "pdf_url": "https://arxiv.org/pdf/2607.01380", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.32014", + "title": "Scalable Behaviour Cloning on Browser Using via Skill Distillation", + "url": "https://arxiv.org/abs/2606.32014", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Kaisen Yang", + "Zheng Jiang", + "Yuzhao Peng", + "Houde Qian", + "Boshi Zhang", + "Youjie Zheng", + "Shijin Hong", + "Qingle Liu", + "Ruoyu Han", + "Bohan Lyu", + "Bingxiang He", + "Eren Cai", + "Calvin Xiao", + "Qinhuai Na" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.32014", + "source": "arxiv", + "source_id": "arxiv:2606.32014", + "pdf_url": "https://arxiv.org/pdf/2606.32014", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.29532", + "title": "SemJoin: Semantic Join Optimization", + "url": "https://arxiv.org/abs/2606.29532", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Christopher Gou", + "Aditya Banerjee", + "Jiaxuan Wang", + "Chunwei Liu" + ], + "categories": [ + "cs.DB", + "cs.AI" + ], + "topics": [ + "agent-evaluation" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29532", + "source": "arxiv", + "source_id": "arxiv:2606.29532", + "pdf_url": "https://arxiv.org/pdf/2606.29532", + "primary_query": "llm-agent" + }, + { + "id": "2606.27951", + "title": "AI Persuasive Framing in Collective Dilemmas", + "url": "https://arxiv.org/abs/2606.27951", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Anders Giovanni Møller", + "Alessia Galdeman", + "Arianna Pera", + "Luca Maria Aiello" + ], + "categories": [ + "cs.CY", + "cs.CL", + "cs.HC", + "physics.soc-ph" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.27951", + "source": "arxiv", + "source_id": "arxiv:2606.27951", + "pdf_url": "https://arxiv.org/pdf/2606.27951", + "primary_query": "ai-agent" + }, + { + "id": "2606.27291", + "title": "Designing Reward Signals for Portable Query Generation: A Case Study in Industrial Semantic Job Search", + "url": "https://arxiv.org/abs/2606.27291", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Ping Liu", + "Qianqi Shen", + "Jianqiang Shen", + "Wenqiong Liu", + "Rajat Arora", + "Yunxiang Ren", + "Chunnan Yao", + "Dan Xu", + "Baofen Zheng", + "Wanjun Jiang", + "Andrii Soviak", + "Kevin Kao", + "Jingwei Wu", + "Wenjing Zhang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.27291", + "source": "arxiv", + "source_id": "arxiv:2606.27291", + "pdf_url": "https://arxiv.org/pdf/2606.27291", + "primary_query": "ai-agent" + }, + { + "id": "2606.25244", + "title": "Reading AI Model Compilation in MLIR Through the Lens of Formal Theories", + "url": "https://arxiv.org/abs/2606.25244", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Javed Absar" + ], + "categories": [ + "cs.PL" + ], + "topics": [ + "coding-agent", + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.25244", + "source": "arxiv", + "source_id": "arxiv:2606.25244", + "pdf_url": "https://arxiv.org/pdf/2606.25244", + "primary_query": "coding-agent" + }, + { + "id": "2606.23768", + "title": "Cryptographic certificates of validity for trustworthy AI", + "url": "https://arxiv.org/abs/2606.23768", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Murdoch J. Gabbay" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LO" + ], + "topics": [ + "reasoning", + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.23768", + "source": "arxiv", + "source_id": "arxiv:2606.23768", + "pdf_url": "https://arxiv.org/pdf/2606.23768", + "primary_query": "agentic-ai" + }, + { + "id": "2606.22447", + "title": "A Differentiable Atari VCS:A Complex, Fully Known Ground Truth for Explainable AI", + "url": "https://arxiv.org/abs/2606.22447", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Andreas Maier", + "Siming Bayer", + "Patrick Krauss" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "coding-agent", + "planning" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.22447", + "source": "arxiv", + "source_id": "arxiv:2606.22447", + "pdf_url": "https://arxiv.org/pdf/2606.22447", + "primary_query": "coding-agent" + }, + { + "id": "2606.19924", + "title": "The Tao of Agency: Autotelic AI, Embedded Agency and Dissolution of the Self", + "url": "https://arxiv.org/abs/2606.19924", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Aritra Sarkar" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.19924", + "source": "arxiv", + "source_id": "arxiv:2606.19924", + "pdf_url": "https://arxiv.org/pdf/2606.19924", + "primary_query": "ai-agent" + }, + { + "id": "2606.19047", + "title": "RODS: Reward-Driven Online Data Synthesis for Multi-Turn Tool-Use Agents", + "url": "https://arxiv.org/abs/2606.19047", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Ruishan Fang", + "Siyuan Lu", + "Chenyi Zhuang", + "Tao Lin" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.19047", + "source": "arxiv", + "source_id": "arxiv:2606.19047", + "pdf_url": "https://arxiv.org/pdf/2606.19047", + "primary_query": "tool-use" + }, + { + "id": "2606.19458", + "title": "MonaVec: A Training-Free Embedded Vector Search Kernel for Edge and Offline AI Systems", + "url": "https://arxiv.org/abs/2606.19458", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Oğuzhan Yenen" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "function-calling", + "rag-agent" + ], + "arxiv_id": "2606.19458", + "source": "arxiv", + "source_id": "arxiv:2606.19458", + "pdf_url": "https://arxiv.org/pdf/2606.19458", + "primary_query": "function-calling" + }, + { + "id": "2606.17321", + "title": "ProCUA-SFT Technical Report", + "url": "https://arxiv.org/abs/2606.17321", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Jaehun Jung", + "Ximing Lu", + "Brandon Cui", + "Muhammad Khalifa", + "Shaokun Zhang", + "Hao Zhang", + "Jin Xu", + "Amala Sanjay Deshmukh", + "Karan Sapra", + "Andrew Tao", + "Yejin Choi", + "Jan Kautz", + "Mingjie Liu", + "Yi Dong" + ], + "categories": [ + "cs.LG", + "cs.CV" + ], + "topics": [ + "computer-use", + "planning", + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.17321", + "source": "arxiv", + "source_id": "arxiv:2606.17321", + "pdf_url": "https://arxiv.org/pdf/2606.17321", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.15071", + "title": "Quantum learning with a single-atom sensor", + "url": "https://arxiv.org/abs/2606.15071", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Yin Mo", + "Emilio Bagan", + "Giulio Chiribella" + ], + "categories": [ + "quant-ph" + ], + "topics": [ + "memory", + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.15071", + "source": "arxiv", + "source_id": "arxiv:2606.15071", + "pdf_url": "https://arxiv.org/pdf/2606.15071", + "primary_query": "agent-memory" + }, + { + "id": "2606.09416", + "title": "Harness Engineering for Physical AI: Robot Middleware Is the Harness Layer", + "url": "https://arxiv.org/abs/2606.09416", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Sanghoon Lee", + "Jiyeong Chae", + "Kyung-Joon Park" + ], + "categories": [ + "cs.RO", + "cs.AI", + "cs.SE" + ], + "topics": [ + "embodied-agent", + "planning", + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.09416", + "source": "arxiv", + "source_id": "arxiv:2606.09416", + "pdf_url": "https://arxiv.org/pdf/2606.09416", + "primary_query": "language-agent" + }, + { + "id": "2606.02840", + "title": "Self-Regulation through Communication in Evolved Neural Agents", + "url": "https://arxiv.org/abs/2606.02840", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Joshua Nunley" + ], + "categories": [ + "q-bio.PE", + "cs.MA", + "cs.NE", + "nlin.AO" + ], + "topics": [ + "agent-safety" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.02840", + "source": "arxiv", + "source_id": "arxiv:2606.02840", + "pdf_url": "https://arxiv.org/pdf/2606.02840", + "primary_query": "agent-safety" + }, + { + "id": "2606.02528", + "title": "Auditing Asset-Specific Preferences in Financial Large Language Models: Evidence from Bitcoin Representations and Portfolio Allocation", + "url": "https://arxiv.org/abs/2606.02528", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Wenbin Wu" + ], + "categories": [ + "q-fin.GN", + "cs.CY", + "cs.LG" + ], + "topics": [ + "rag" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.02528", + "source": "arxiv", + "source_id": "arxiv:2606.02528", + "pdf_url": "https://arxiv.org/pdf/2606.02528", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.08378", + "title": "Reinforcement Learning for Scalable and Trustworthy Intelligent Systems", + "url": "https://arxiv.org/abs/2605.08378", + "published": "2026-05-08", + "updated": "2026-05-08", + "authors": [ + "Guangchen Lan" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-safety" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.08378", + "source": "arxiv", + "source_id": "arxiv:2605.08378", + "pdf_url": "https://arxiv.org/pdf/2605.08378", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.05795", + "title": "Reward Shaping and Action Masking for Compositional Tasks using Behavior Trees and LLMs", + "url": "https://arxiv.org/abs/2605.05795", + "published": "2026-05-07", + "updated": "2026-05-23", + "authors": [ + "Nicholas Potteiger", + "Ankita Samaddar", + "Taylor T. Johnson", + "Xenofon Koutsoukos" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.05795", + "source": "arxiv", + "source_id": "arxiv:2605.05795", + "pdf_url": "https://arxiv.org/pdf/2605.05795", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2509.25378", + "title": "Detecting and Fixing API Misuses of Data Science Libraries Using Large Language Models", + "url": "https://arxiv.org/abs/2509.25378", + "published": "2025-09-29", + "updated": "2025-09-29", + "authors": [ + "Akalanka Galappaththi", + "Francisco Ribeiro", + "Sarah Nadi" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.25378", + "source": "arxiv", + "source_id": "arxiv:2509.25378", + "pdf_url": "https://arxiv.org/pdf/2509.25378", + "primary_query": "function-calling" + }, + { + "id": "2509.06736", + "title": "VehicleWorld: A Highly Integrated Multi-Device Environment for Intelligent Vehicle Interaction", + "url": "https://arxiv.org/abs/2509.06736", + "published": "2025-09-08", + "updated": "2025-09-08", + "authors": [ + "Jie Yang", + "Jiajun Chen", + "Zhangyue Yin", + "Shuo Chen", + "Yuxin Wang", + "Yiran Guo", + "Yuan Li", + "Yining Zheng", + "Xuanjing Huang", + "Xipeng Qiu" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 6, + "relevance": "low", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.06736", + "source": "arxiv", + "source_id": "arxiv:2509.06736", + "pdf_url": "https://arxiv.org/pdf/2509.06736", + "primary_query": "function-calling" + }, + { + "id": "2607.01188", + "title": "Optimal Resource Utilization for Autonomous Laboratory Orchestrators", + "url": "https://arxiv.org/abs/2607.01188", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Austin McDannald", + "Julia Tisaranni", + "Howie Joress" + ], + "categories": [ + "cs.AI", + "cond-mat.mtrl-sci" + ], + "topics": [ + "planning" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.01188", + "source": "arxiv", + "source_id": "arxiv:2607.01188", + "pdf_url": "https://arxiv.org/pdf/2607.01188", + "primary_query": "ai-agent" + }, + { + "id": "2606.31132", + "title": "ELASTIC: Efficiently Learning to Adaptively Scale Test-Time Compute for Generative Control Policies", + "url": "https://arxiv.org/abs/2606.31132", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Andrew Zou Li", + "Gokul Swamy", + "Yonatan Bisk", + "Andrea Bajcsy" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "tool-use" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.31132", + "source": "arxiv", + "source_id": "arxiv:2606.31132", + "pdf_url": "https://arxiv.org/pdf/2606.31132", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.30543", + "title": "TRACE: Temporal Relationship-Aware Conversational Entrainment Detection in Dyadic Speech", + "url": "https://arxiv.org/abs/2606.30543", + "published": "2026-06-29", + "updated": "2026-07-03", + "authors": [ + "Sathvik Manikantan Napa Ugandhar", + "Hao Zhang", + "Alison Gunzler", + "Yuzhe Wang", + "Thomas Thebaud", + "Georgi Tinchev", + "Venkatesh Ravichandran", + "Laureano Moro-Velázquez" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "tool-use" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.30543", + "source": "arxiv", + "source_id": "arxiv:2606.30543", + "pdf_url": "https://arxiv.org/pdf/2606.30543", + "primary_query": "ai-agent" + }, + { + "id": "2606.27045", + "title": "The Spec Growth Engine: Spec-Anchored, Code-Coupled, Drift-Enforced Architecture for AI-Assisted Software Development", + "url": "https://arxiv.org/abs/2606.27045", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Hartwig Grabowski" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.27045", + "source": "arxiv", + "source_id": "arxiv:2606.27045", + "pdf_url": "https://arxiv.org/pdf/2606.27045", + "primary_query": "coding-agent" + }, + { + "id": "2606.27365", + "title": "3D Imaging of Complex Skyrmion and Hopf Topologies in an Extended Sample", + "url": "https://arxiv.org/abs/2606.27365", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "I. Binnie", + "H. Fang", + "B. Shearer", + "A. Grafov", + "N. Jenkins", + "Y. Shao", + "C. O'Leary", + "Y. Liao", + "T. Feggeler", + "A. Oh", + "S. Yazdi", + "J. Zou", + "B. Wang", + "E-E. Cating", + "S. A. Montoya", + "D. Shapiro", + "J. Miao", + "H. C. Kapteyn", + "M. M. Murnane" + ], + "categories": [ + "cond-mat.mtrl-sci", + "cond-mat.mes-hall", + "physics.app-ph" + ], + "topics": [ + "memory", + "rag", + "tool-use" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.27365", + "source": "arxiv", + "source_id": "arxiv:2606.27365", + "pdf_url": "https://arxiv.org/pdf/2606.27365", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.25337", + "title": "AI Coaching for Accelerating Human Skill Development with Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.25337", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Wei Wang", + "Enlin Gu", + "Antonio Loquercio", + "Haimin Hu", + "Rahul Mangharam" + ], + "categories": [ + "cs.RO", + "cs.AI", + "cs.HC" + ], + "topics": [ + "embodied-agent" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.25337", + "source": "arxiv", + "source_id": "arxiv:2606.25337", + "pdf_url": "https://arxiv.org/pdf/2606.25337", + "primary_query": "ai-agent" + }, + { + "id": "2606.23315", + "title": "Test-Driven, AI-Assisted Learning: Replacing Lectures with Weekly Closed-Book Tests", + "url": "https://arxiv.org/abs/2606.23315", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Jin-Guo Liu", + "Shang-Qi Lu", + "Xin-Ran Shi", + "Long-Li Zheng", + "Wei Wang" + ], + "categories": [ + "cs.CY" + ], + "topics": [ + "workflow-agent" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.23315", + "source": "arxiv", + "source_id": "arxiv:2606.23315", + "pdf_url": "https://arxiv.org/pdf/2606.23315", + "primary_query": "ai-agent" + }, + { + "id": "2606.22568", + "title": "SeFi-Image: A Text-to-Image Foundation Model with Semantic-First Diffusion", + "url": "https://arxiv.org/abs/2606.22568", + "published": "2026-06-21", + "updated": "2026-06-26", + "authors": [ + "SeFi-Team" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.22568", + "source": "arxiv", + "source_id": "arxiv:2606.22568", + "pdf_url": "https://arxiv.org/pdf/2606.22568", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.22226", + "title": "Quantifying Theoretical AI Alignment Guarantees: Receiver-Utility Bounds in Bayesian Persuasion", + "url": "https://arxiv.org/abs/2606.22226", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Eric Yachbes", + "Eva Tardos" + ], + "categories": [ + "cs.GT", + "cs.AI", + "cs.IT" + ], + "topics": [ + "agent-safety" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.22226", + "source": "arxiv", + "source_id": "arxiv:2606.22226", + "pdf_url": "https://arxiv.org/pdf/2606.22226", + "primary_query": "ai-agent" + }, + { + "id": "2606.19683", + "title": "Exit-and-Join Dynamics for Decentralized Coalition Formation", + "url": "https://arxiv.org/abs/2606.19683", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Quanyan Zhu" + ], + "categories": [ + "cs.AI", + "cs.MA", + "eess.SY" + ], + "topics": [ + "agent-evaluation" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.19683", + "source": "arxiv", + "source_id": "arxiv:2606.19683", + "pdf_url": "https://arxiv.org/pdf/2606.19683", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.17790", + "title": "Distributed Experimental Design: Bayes-optimal Fusion of Local Designs", + "url": "https://arxiv.org/abs/2606.17790", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Nagananda K G", + "Lav R. Varshney", + "Pramod K. Varshney" + ], + "categories": [ + "stat.AP", + "cs.IT" + ], + "topics": [ + "agent-evaluation", + "planning" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.17790", + "source": "arxiv", + "source_id": "arxiv:2606.17790", + "pdf_url": "https://arxiv.org/pdf/2606.17790", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.17388", + "title": "Agent Utilities over Generalized Voronoi Regions and their Gradients", + "url": "https://arxiv.org/abs/2606.17388", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Andre N. Costa", + "Petter Ögren", + "Carlos H. C. Ribeiro" + ], + "categories": [ + "cs.RO", + "cs.CG", + "eess.SY" + ], + "topics": [ + "agent" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.17388", + "source": "arxiv", + "source_id": "arxiv:2606.17388", + "pdf_url": "https://arxiv.org/pdf/2606.17388", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.12235", + "title": "BenDi: An Energy-Efficient Quasi-Stochastic Systolic Architecture for Edge Bioelectronics", + "url": "https://arxiv.org/abs/2606.12235", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Bochen Ye", + "Yihan Pan", + "Shady Agwa", + "Themis Prodromakis" + ], + "categories": [ + "cs.AR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.12235", + "source": "arxiv", + "source_id": "arxiv:2606.12235", + "pdf_url": "https://arxiv.org/pdf/2606.12235", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.04131", + "title": "Network node immunization: improving Netshield algorithm through random rooted forests", + "url": "https://arxiv.org/abs/2606.04131", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Luca Avena", + "Alexandre Gaudillière", + "Irina Gurewitsch", + "Adoré Randriamandroso", + "Alessio Troiani" + ], + "categories": [ + "cs.SI", + "math.PR" + ], + "topics": [ + "agent-evaluation" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2606.04131", + "source": "arxiv", + "source_id": "arxiv:2606.04131", + "pdf_url": "https://arxiv.org/pdf/2606.04131", + "primary_query": "function-calling" + }, + { + "id": "2605.21792", + "title": "Residual Skill Optimization for Text-to-SQL Ensembles", + "url": "https://arxiv.org/abs/2605.21792", + "published": "2026-05-20", + "updated": "2026-05-20", + "authors": [ + "Jiongli Zhu", + "Haoquan Guan", + "Parjanya Prajakta Prashant", + "Nikki Lijing Kuang", + "Seyedeh Baharan Khatami", + "Canwen Xu", + "Xiaodong Yu", + "Yingyu Lin", + "Zhewei Yao", + "Yuxiong He", + "Babak Salimi" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.DB", + "cs.LG" + ], + "topics": [ + "coding-agent" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.21792", + "source": "arxiv", + "source_id": "arxiv:2605.21792", + "pdf_url": "https://arxiv.org/pdf/2605.21792", + "primary_query": "function-calling" + }, + { + "id": "2602.23397", + "title": "Lifecycle-Integrated Security for AI-Cloud Convergence in Cyber-Physical Infrastructure", + "url": "https://arxiv.org/abs/2602.23397", + "published": "2026-02-26", + "updated": "2026-02-26", + "authors": [ + "S M Zia Ur Rashid", + "Deepa Gurung", + "Sonam Raj Gupta", + "Suman Rath" + ], + "categories": [ + "cs.CR", + "eess.SY" + ], + "topics": [ + "agent-safety" + ], + "score": 5, + "relevance": "low", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.23397", + "source": "arxiv", + "source_id": "arxiv:2602.23397", + "pdf_url": "https://arxiv.org/pdf/2602.23397", + "primary_query": "agent-safety" + }, + { + "id": "2607.05498", + "title": "Non-spherical Cows: Introducing the Asphericity Parameter as a Measure of Accretion Geometry", + "url": "https://arxiv.org/abs/2607.05498", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Benjamin A. Seidel", + "Rhea-Silvia Remus", + "Lucas C. Kimmig", + "Lucas M. Valenzuela", + "Klaus Dolag" + ], + "categories": [ + "astro-ph.GA", + "astro-ph.CO" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.05498", + "source": "arxiv", + "source_id": "arxiv:2607.05498", + "pdf_url": "https://arxiv.org/pdf/2607.05498", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.04222", + "title": "Unsupervised Features Mining via Activation Geometry", + "url": "https://arxiv.org/abs/2607.04222", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Amit LeVi", + "Elad David", + "Max Fomin" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "reasoning" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04222", + "source": "arxiv", + "source_id": "arxiv:2607.04222", + "pdf_url": "https://arxiv.org/pdf/2607.04222", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03955", + "title": "Strategy-Proof Probabilistic Social Choice Correspondences under Conditional Expected Utility", + "url": "https://arxiv.org/abs/2607.03955", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Madhuparna Karmokar", + "Ujjwal Kumar", + "Soumyarup Sadhukhan" + ], + "categories": [ + "econ.TH" + ], + "topics": [ + "agent-evaluation" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2607.03955", + "source": "arxiv", + "source_id": "arxiv:2607.03955", + "pdf_url": "https://arxiv.org/pdf/2607.03955", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.03949", + "title": "TESSERA v2: Scaling Pixel-wise Earth Foundation Models", + "url": "https://arxiv.org/abs/2607.03949", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Zhengpeng Feng", + "Sadiq Jaffer", + "Ira Shokar", + "Jovana Knezevic", + "Mark Elvers", + "Clement Atzberger", + "Robin Young", + "Aneesh Naik", + "Niall Robinson", + "Andrew Blake", + "David Coomes", + "Anil Madhavapeddy", + "Srinivasan Keshav" + ], + "categories": [ + "cs.CV", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.03949", + "source": "arxiv", + "source_id": "arxiv:2607.03949", + "pdf_url": "https://arxiv.org/pdf/2607.03949", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.05439", + "title": "Design-CP: Context Parallelism for Design of Protein Nanoparticles", + "url": "https://arxiv.org/abs/2607.05439", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Lorenzo Tarricone", + "Helen E. Eisenach", + "Aiko Muraishi", + "Charlotte M. Deane" + ], + "categories": [ + "cs.LG", + "cs.DC", + "q-bio.QM" + ], + "topics": [ + "agent-evaluation", + "memory" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.05439", + "source": "arxiv", + "source_id": "arxiv:2607.05439", + "pdf_url": "https://arxiv.org/pdf/2607.05439", + "primary_query": "agentic-ai" + }, + { + "id": "2606.28801", + "title": "Cross-channel Specific Emitter Identification and Verification via Signal Envelope", + "url": "https://arxiv.org/abs/2606.28801", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Yuhao Chen", + "Boxiang He", + "Shilian Wang", + "Jing Lei" + ], + "categories": [ + "eess.SP" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.28801", + "source": "arxiv", + "source_id": "arxiv:2606.28801", + "pdf_url": "https://arxiv.org/pdf/2606.28801", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.22997", + "title": "A Greatest Common Divisor Criterion of Certain Binomial Coefficients", + "url": "https://arxiv.org/abs/2606.22997", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Dakai Guo", + "Ruichen Qiu", + "Yichuan Cao", + "Ruyong Feng", + "Xiao-Shan Gao" + ], + "categories": [ + "math.NT", + "cs.LO" + ], + "topics": [ + "agent" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.22997", + "source": "arxiv", + "source_id": "arxiv:2606.22997", + "pdf_url": "https://arxiv.org/pdf/2606.22997", + "primary_query": "ai-agent" + }, + { + "id": "2606.20208", + "title": "Beyond Accuracy: Measuring Logical Compliance of Predictive Models", + "url": "https://arxiv.org/abs/2606.20208", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Guillaume Olivier Delplanque", + "Pierre Genevès", + "Nabil Layaïda", + "Zephirin Faure" + ], + "categories": [ + "cs.AI", + "cs.DB", + "cs.NE" + ], + "topics": [ + "agent-evaluation" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20208", + "source": "arxiv", + "source_id": "arxiv:2606.20208", + "pdf_url": "https://arxiv.org/pdf/2606.20208", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.23719", + "title": "A Hybrid Quantum-Classical Approach for Melt Pool Prediction in Laser Powder Bed Fusion", + "url": "https://arxiv.org/abs/2606.23719", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Matthew M. Sato", + "Kincho H. Law" + ], + "categories": [ + "quant-ph", + "cs.LG" + ], + "topics": [ + "coding-agent", + "rag", + "tool-use" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.23719", + "source": "arxiv", + "source_id": "arxiv:2606.23719", + "pdf_url": "https://arxiv.org/pdf/2606.23719", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.17860", + "title": "An Epistemic Analysis of Random Coordinated Attack", + "url": "https://arxiv.org/abs/2606.17860", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Sophia Knight", + "David Lehnherr", + "Sergio Rajsbaum" + ], + "categories": [ + "cs.DC", + "cs.LO" + ], + "topics": [ + "computer-use", + "reasoning", + "tool-use" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.17860", + "source": "arxiv", + "source_id": "arxiv:2606.17860", + "pdf_url": "https://arxiv.org/pdf/2606.17860", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13225", + "title": "The QR Factorization for Banded-Plus-Semiseparable Matrices Is Computable in Linear Complexity", + "url": "https://arxiv.org/abs/2606.13225", + "published": "2026-06-11", + "updated": "2026-06-12", + "authors": [ + "Tao Chen", + "Sheehan Olver" + ], + "categories": [ + "math.NA" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.13225", + "source": "arxiv", + "source_id": "arxiv:2606.13225", + "pdf_url": "https://arxiv.org/pdf/2606.13225", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.07463", + "title": "Amortized Neural Optimization for Pre-Layout Signal Integrity Design Space Exploration using Differentiable Surrogates", + "url": "https://arxiv.org/abs/2606.07463", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Julian Withöft", + "Werner John", + "Emre Ecik", + "Ralf Brüning", + "Jürgen Götze" + ], + "categories": [ + "eess.SP", + "cs.CE", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "workflow-agent", + "world-model" + ], + "score": 4, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.07463", + "source": "arxiv", + "source_id": "arxiv:2606.07463", + "pdf_url": "https://arxiv.org/pdf/2606.07463", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.31582", + "title": "Generalized Laura-Andoyer equations and the enumeration of some symmetrical classes of Dziobek configurations", + "url": "https://arxiv.org/abs/2606.31582", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Thiago Dias", + "Ya-Lun Tsai" + ], + "categories": [ + "math.DS" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.31582", + "source": "arxiv", + "source_id": "arxiv:2606.31582", + "pdf_url": "https://arxiv.org/pdf/2606.31582", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.30536", + "title": "Evaluating the Fourier Approximation in Pulsar Timing Array Analysis", + "url": "https://arxiv.org/abs/2606.30536", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yongqi Zhang", + "Hayden Scholz", + "Ken D. Olum", + "Lucas Steinberger", + "Gabriella Agazie", + "Akash Anumarlapudi", + "Anne M. Archibald", + "Zaven Arzoumanian", + "Paul T. Baker", + "Paul R. Brook", + "H. Thankful Cromartie", + "Kathryn Crowter", + "Megan E. DeCesar", + "Paul B. Demorest", + "Timothy Dolch", + "Justin A. Ellis", + "Elizabeth C. Ferrara", + "William Fiore", + "Emmanuel Fonseca", + "Gabriel E. Freedman", + "Nate Garver-Daniels", + "Peter A. Gentile", + "Joseph Glaser", + "Deborah C. Good", + "Jeffrey S. Hazboun", + "Ross J. Jennings", + "Megan L. Jones", + "David L. Kaplan", + "Matthew Kerr", + "Michael T. Lam", + "Duncan R. Lorimer", + "Jing Luo", + "Ryan S. Lynch", + "Alexander McEwen", + "Maura A. McLaughlin", + "Natasha McMann", + "Bradley W. Meyers", + "Cherry Ng", + "David J. Nice", + "Timothy T. Pennucci", + "Benetge B. P. Perera", + "Nihan S. Pol", + "Henri A. Radovan", + "Scott M. Ransom", + "Paul S. Ray", + "Ann Schmiedekamp", + "Carl Schmiedekamp", + "Brent J. Shapiro-Albert", + "Ingrid H. Stairs", + "Kevin Stovall", + "Abhimanyu Susobhanan", + "Joseph K. Swiggum", + "Stephen R. Taylor", + "Michele Vallisneri", + "Rutger van Haasteren", + "Haley M. Wahl" + ], + "categories": [ + "gr-qc", + "astro-ph.HE" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.30536", + "source": "arxiv", + "source_id": "arxiv:2606.30536", + "pdf_url": "https://arxiv.org/pdf/2606.30536", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.23478", + "title": "ffortissimo: A Freeform Forward-Modeling Pipeline for High-Contrast Images of Circumstellar Disks Based on Automatic Differentiation", + "url": "https://arxiv.org/abs/2606.23478", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Jay K. Kueny", + "Joseph D. Long", + "Jared R. Males", + "Alycia J. Weinberger", + "Laird M. Close", + "Joshua Liberman", + "Sebastiaan Haffert", + "Eden McEwen", + "Maggie Y. Kautz", + "Olivier Guyon", + "Logan Pearce", + "Parker T. Johnson", + "Katie Twitchell", + "Jialin Li", + "Alex Hedglen", + "Avalon Gower", + "Warren Foster", + "Jhen Lumbres", + "Lauren Schatz" + ], + "categories": [ + "astro-ph.IM", + "astro-ph.SR" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.23478", + "source": "arxiv", + "source_id": "arxiv:2606.23478", + "pdf_url": "https://arxiv.org/pdf/2606.23478", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.20257", + "title": "Measurements of charged-particle pseudorapidity and transverse momentum distributions in O+O and Ne+Ne collisions at $\\sqrt{s_{_\\text{NN}}} = 5.36$ TeV with the ATLAS detector", + "url": "https://arxiv.org/abs/2606.20257", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "ATLAS Collaboration" + ], + "categories": [ + "nucl-ex", + "hep-ex" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20257", + "source": "arxiv", + "source_id": "arxiv:2606.20257", + "pdf_url": "https://arxiv.org/pdf/2606.20257", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.18975", + "title": "On the robustness of the angular homogeneity scale $θ_H$: a comparative analysis of computational approaches", + "url": "https://arxiv.org/abs/2606.18975", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Pedro Fanha", + "António da Silva", + "José Fonseca", + "José Pedro Mimoso", + "Ziad Sakr" + ], + "categories": [ + "astro-ph.CO" + ], + "topics": [ + "agent-evaluation", + "world-model" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.18975", + "source": "arxiv", + "source_id": "arxiv:2606.18975", + "pdf_url": "https://arxiv.org/pdf/2606.18975", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.15776", + "title": "Spin-dependent electron transfer through a ring-wire coupled junction: Role of in-plane electric field", + "url": "https://arxiv.org/abs/2606.15776", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Prabhab Patra", + "Santanu K. Maiti" + ], + "categories": [ + "cond-mat.mes-hall" + ], + "topics": [ + "computer-use", + "planning" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.15776", + "source": "arxiv", + "source_id": "arxiv:2606.15776", + "pdf_url": "https://arxiv.org/pdf/2606.15776", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.15336", + "title": "Nonlocal Orbital-Free Kinetic Energy Functional from the Jellium-with-Gap Model for Finite Systems", + "url": "https://arxiv.org/abs/2606.15336", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Abhishek Bhattacharjee", + "Subrata Jana", + "Szymon Smiga", + "Prasanjit Samal" + ], + "categories": [ + "cond-mat.mtrl-sci" + ], + "topics": [ + "agent-evaluation" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.15336", + "source": "arxiv", + "source_id": "arxiv:2606.15336", + "pdf_url": "https://arxiv.org/pdf/2606.15336", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.14131", + "title": "G-computation for causal effect estimation from observational hierarchical data with unmeasured cluster context", + "url": "https://arxiv.org/abs/2606.14131", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Shafayet Khan Shafee", + "Bishal Sarker", + "Md. Niamul Islam Sium" + ], + "categories": [ + "stat.ME" + ], + "topics": [ + "agent-evaluation", + "world-model" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.14131", + "source": "arxiv", + "source_id": "arxiv:2606.14131", + "pdf_url": "https://arxiv.org/pdf/2606.14131", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13980", + "title": "On Cutting Cakes and Crossing Curves", + "url": "https://arxiv.org/abs/2606.13980", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Alexandros Hollender", + "Gilbert Maystre", + "Kilian Risse" + ], + "categories": [ + "cs.GT", + "cs.CC" + ], + "topics": [ + "agent" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.13980", + "source": "arxiv", + "source_id": "arxiv:2606.13980", + "pdf_url": "https://arxiv.org/pdf/2606.13980", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.08070", + "title": "Earth-Density Stratification and Quantum Gravity Corrections in Long-Baseline Neutrino Oscillation Experiments", + "url": "https://arxiv.org/abs/2606.08070", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Bipin Singh Koranga", + "Vivek Kumar Nautiya" + ], + "categories": [ + "hep-ph" + ], + "topics": [ + "agent-evaluation", + "planning" + ], + "score": 3, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.08070", + "source": "arxiv", + "source_id": "arxiv:2606.08070", + "pdf_url": "https://arxiv.org/pdf/2606.08070", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.01374", + "title": "Algebraic conditions for second-moment stability boundaries of linear, time-invariant stochastic delay-differential equations", + "url": "https://arxiv.org/abs/2607.01374", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Zsolt Iklodi", + "Harry Dankowicz" + ], + "categories": [ + "math.DS" + ], + "topics": [ + "world-model" + ], + "score": 2, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.01374", + "source": "arxiv", + "source_id": "arxiv:2607.01374", + "pdf_url": "https://arxiv.org/pdf/2607.01374", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.30540", + "title": "Synthesizability and Mechanical Properties of High-Entropy Borides: First-Principles and Machine Learning Studies", + "url": "https://arxiv.org/abs/2606.30540", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Luke Moore", + "Ethan Fox", + "Bria Storr", + "Jayden R. Palomino", + "Shane A. Catledge", + "Yogesh K. Vohra", + "Cheng-Chien Chen" + ], + "categories": [ + "cond-mat.mtrl-sci" + ], + "topics": [ + "agent-evaluation" + ], + "score": 2, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.30540", + "source": "arxiv", + "source_id": "arxiv:2606.30540", + "pdf_url": "https://arxiv.org/pdf/2606.30540", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.29989", + "title": "Rendering Coherent Scattering via Quantum Collision Models", + "url": "https://arxiv.org/abs/2606.29989", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "João S. Ferreira", + "Spencer S. Topel", + "Pierre Fromholz", + "James R. Wootton" + ], + "categories": [ + "cs.GR", + "physics.pop-ph", + "quant-ph" + ], + "topics": [ + "tool-use" + ], + "score": 2, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.29989", + "source": "arxiv", + "source_id": "arxiv:2606.29989", + "pdf_url": "https://arxiv.org/pdf/2606.29989", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.28146", + "title": "A statistically robust framework for detecting and classifying hysteresis patterns in astrophysical spectral evolution", + "url": "https://arxiv.org/abs/2606.28146", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Tomislav Terzić" + ], + "categories": [ + "astro-ph.HE", + "astro-ph.IM" + ], + "topics": [ + "planning" + ], + "score": 2, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.28146", + "source": "arxiv", + "source_id": "arxiv:2606.28146", + "pdf_url": "https://arxiv.org/pdf/2606.28146", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.20975", + "title": "Solving Einstein Field Equations on a Digital Quantum Computer", + "url": "https://arxiv.org/abs/2606.20975", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Clelia Altomonte", + "Malcolm Fairbairn" + ], + "categories": [ + "gr-qc" + ], + "topics": [ + "world-model" + ], + "score": 2, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20975", + "source": "arxiv", + "source_id": "arxiv:2606.20975", + "pdf_url": "https://arxiv.org/pdf/2606.20975", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.20796", + "title": "$N=1$ Supersymmetry, Weil-Petersson Volume Recursion, and a Spectral Curve", + "url": "https://arxiv.org/abs/2606.20796", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Clifford V. Johnson" + ], + "categories": [ + "hep-th", + "math-ph" + ], + "topics": [ + "agent-evaluation" + ], + "score": 2, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20796", + "source": "arxiv", + "source_id": "arxiv:2606.20796", + "pdf_url": "https://arxiv.org/pdf/2606.20796", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.18196", + "title": "Receiver-Aware Analysis and Verification of the Spectral Separation Coefficient Under Interference-Induced Degradation", + "url": "https://arxiv.org/abs/2606.18196", + "published": "2026-06-16", + "updated": "2026-06-18", + "authors": [ + "Lucas Heublein", + "Fabian Benschuh", + "Alexander Rügamer", + "Felix Ott" + ], + "categories": [ + "eess.SP" + ], + "topics": [ + "agent-evaluation" + ], + "score": 2, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.18196", + "source": "arxiv", + "source_id": "arxiv:2606.18196", + "pdf_url": "https://arxiv.org/pdf/2606.18196", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.12698", + "title": "Higher Dimensional Loop Quantum Black hole in de Sitter Spacetime: Quasinormal Modes and Shadow Signatures", + "url": "https://arxiv.org/abs/2606.12698", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Kourosh Nozari", + "Sara Saghafi", + "Ali Mohammadpour" + ], + "categories": [ + "gr-qc", + "hep-ph", + "hep-th" + ], + "topics": [ + "tool-use" + ], + "score": 2, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.12698", + "source": "arxiv", + "source_id": "arxiv:2606.12698", + "pdf_url": "https://arxiv.org/pdf/2606.12698", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.10341", + "title": "Global polarization of $Λ$, $Ξ^{-}$, and $Ω^{-}$ hyperons in Au+Au collisions at RHIC BES-II energies", + "url": "https://arxiv.org/abs/2606.10341", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Gen-Hui Li", + "Cong Yi", + "Xiang-Yu Wu", + "Shi Pu", + "Guang-You Qin" + ], + "categories": [ + "nucl-th" + ], + "topics": [ + "tool-use" + ], + "score": 2, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.10341", + "source": "arxiv", + "source_id": "arxiv:2606.10341", + "pdf_url": "https://arxiv.org/pdf/2606.10341", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.02160", + "title": "Algorithms for hyperelliptic Mumford Curves $p$-adic Uniformization, $p$-adic integrals and $p$-adic heights", + "url": "https://arxiv.org/abs/2607.02160", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Enis Kaya", + "Marc Masdeu", + "J. Steffen Müller", + "Marius van der Put" + ], + "categories": [ + "math.NT", + "math.AG" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.02160", + "source": "arxiv", + "source_id": "arxiv:2607.02160", + "pdf_url": "https://arxiv.org/pdf/2607.02160", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.01650", + "title": "Computed emissivity of carbon dioxide, water vapor, and their mixtures for a wide range of temperatures and pressure-pathlengths", + "url": "https://arxiv.org/abs/2607.01650", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Osama A. Marzouk" + ], + "categories": [ + "physics.gen-ph" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.01650", + "source": "arxiv", + "source_id": "arxiv:2607.01650", + "pdf_url": "https://arxiv.org/pdf/2607.01650", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.00476", + "title": "Complexity of Low-Degree Skew Polynomial Multiplication over Finite Fields", + "url": "https://arxiv.org/abs/2607.00476", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Ke Ye", + "Yichuan Cao", + "Ruichen Qiu" + ], + "categories": [ + "cs.SC" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.00476", + "source": "arxiv", + "source_id": "arxiv:2607.00476", + "pdf_url": "https://arxiv.org/pdf/2607.00476", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.29294", + "title": "Quantum models of the Riemann zeta function, lattice spin models and algebraic models of entanglement", + "url": "https://arxiv.org/abs/2606.29294", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Nikolaj M. Glazunov" + ], + "categories": [ + "math.NT" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.29294", + "source": "arxiv", + "source_id": "arxiv:2606.29294", + "pdf_url": "https://arxiv.org/pdf/2606.29294", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.27415", + "title": "Elimination of Flux Trapping in Superconducting Circuits in Ambient Magnetic Fields", + "url": "https://arxiv.org/abs/2606.27415", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Rohan T. Kapur", + "Alex Wynn", + "Sergey K. Tolpygo", + "Neel Parmar", + "Anil Mankame", + "Adam A. Libson", + "Rabindra Das", + "Michele Kelley", + "Pauli Kehayias", + "Nathaniel J. O'Connor", + "Collin N. Muniz", + "Justin L. Mallek", + "Jennifer M. Schloss" + ], + "categories": [ + "cond-mat.supr-con", + "cond-mat.mes-hall", + "physics.app-ph", + "quant-ph" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.27415", + "source": "arxiv", + "source_id": "arxiv:2606.27415", + "pdf_url": "https://arxiv.org/pdf/2606.27415", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.26255", + "title": "Hochschild (co)homology and cyclic homology via a graded Euler characteristic with applications to higher preprojective algebras", + "url": "https://arxiv.org/abs/2606.26255", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Jon Wallem Anundsen", + "Mads Hustad Sandøy" + ], + "categories": [ + "math.RT" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.26255", + "source": "arxiv", + "source_id": "arxiv:2606.26255", + "pdf_url": "https://arxiv.org/pdf/2606.26255", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.21829", + "title": "Elementary solutions of ordinary tropical differential equations, and vanishing orders of solutions of algebraic differential equations", + "url": "https://arxiv.org/abs/2606.21829", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Cristhian Garay-López", + "Johana Luviano-Flores", + "Carla Valencia-Negrete" + ], + "categories": [ + "math.AG", + "math.CO" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.21829", + "source": "arxiv", + "source_id": "arxiv:2606.21829", + "pdf_url": "https://arxiv.org/pdf/2606.21829", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.19503", + "title": "Kernel transformations and bounds for smeared spectral functions", + "url": "https://arxiv.org/abs/2606.19503", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "William I. Jay", + "Matteo Saccardi" + ], + "categories": [ + "hep-lat" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.19503", + "source": "arxiv", + "source_id": "arxiv:2606.19503", + "pdf_url": "https://arxiv.org/pdf/2606.19503", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.16510", + "title": "Petrov-Galerkin Variational Physics-Informed Neural Network Framework for Two-Dimensional Singularly Perturbed Problems", + "url": "https://arxiv.org/abs/2606.16510", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Vijay Kumar", + "Gautam Singh" + ], + "categories": [ + "math.NA", + "cs.LG" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.16510", + "source": "arxiv", + "source_id": "arxiv:2606.16510", + "pdf_url": "https://arxiv.org/pdf/2606.16510", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.12998", + "title": "Efficient emulation of nuclear ground states with neural-network variational Monte Carlo and eigenvector continuation", + "url": "https://arxiv.org/abs/2606.12998", + "published": "2026-06-11", + "updated": "2026-06-12", + "authors": [ + "Mao Li", + "Yilong Yang", + "Pengwei Zhao" + ], + "categories": [ + "nucl-th" + ], + "topics": [], + "score": 1, + "relevance": "low", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.12998", + "source": "arxiv", + "source_id": "arxiv:2606.12998", + "pdf_url": "https://arxiv.org/pdf/2606.12998", + "primary_query": "web-gui-agent" + } +] diff --git a/data/arxiv-agent-papers-2025-07-08-to-2026-07-08.json b/data/arxiv-agent-papers-2025-07-08-to-2026-07-08.json new file mode 100644 index 0000000..3a51822 --- /dev/null +++ b/data/arxiv-agent-papers-2025-07-08-to-2026-07-08.json @@ -0,0 +1,35331 @@ +[ + { + "id": "2606.08340", + "title": "Benchmarking Open-Ended Multi-Agent Coordination in Language Agents", + "url": "https://arxiv.org/abs/2606.08340", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Kale-ab Abebe Tessera", + "Andras Szecsenyi", + "Cameron Barker", + "Alexander Rutherford", + "Davide Paglieri", + "Aidan Scannell", + "Henry Gouk", + "Elliot J. Crowley", + "Tim Rocktäschel", + "Amos Storkey" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 28, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "language-agent", + "planning-agent" + ], + "arxiv_id": "2606.08340", + "source": "arxiv", + "source_id": "arxiv:2606.08340", + "pdf_url": "https://arxiv.org/pdf/2606.08340", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.20833", + "title": "MemGym: a Long-Horizon Memory Environment for LLM Agents", + "url": "https://arxiv.org/abs/2605.20833", + "published": "2026-05-20", + "updated": "2026-05-20", + "authors": [ + "Wujiang Xu", + "Yu Wang", + "Kai Mei", + "Kaiqu Liang", + "Zhenting Wang", + "Mingyu Jin", + "Han Zhang", + "Shi-Xiong Zhang", + "Wenyue Hua", + "Sambit Sahu", + "Dimitris N. Metaxas" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 26, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.20833", + "source": "arxiv", + "source_id": "arxiv:2605.20833", + "pdf_url": "https://arxiv.org/pdf/2605.20833", + "primary_query": "agent-memory" + }, + { + "id": "2607.06008", + "title": "PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2607.06008", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Hongliang Li", + "Yijin Liu", + "Zhiwei Zhang", + "Zihe Liu", + "Xinyue Lou", + "Jinan Xu", + "Fandong Meng", + "Kaiyu Huang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 25, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2607.06008", + "source": "arxiv", + "source_id": "arxiv:2607.06008", + "pdf_url": "https://arxiv.org/pdf/2607.06008", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.28425", + "title": "Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems", + "url": "https://arxiv.org/abs/2606.28425", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Jimmy Laurence Rippin", + "Simon C. Marshall", + "David Demitri Africa", + "Christian Schroeder de Witt" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use" + ], + "score": 25, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "autonomous-agent-llm", + "multi-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.28425", + "source": "arxiv", + "source_id": "arxiv:2606.28425", + "pdf_url": "https://arxiv.org/pdf/2606.28425", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24937", + "title": "The Hitchhiker's Guide to Agentic AI: From Foundations to Systems", + "url": "https://arxiv.org/abs/2606.24937", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Haggai Roitman" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.IR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 25, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm", + "rag-agent", + "tool-use" + ], + "arxiv_id": "2606.24937", + "source": "arxiv", + "source_id": "arxiv:2606.24937", + "pdf_url": "https://arxiv.org/pdf/2606.24937", + "primary_query": "agentic-ai" + }, + { + "id": "2606.10749", + "title": "Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation", + "url": "https://arxiv.org/abs/2606.10749", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yuchen Ling", + "Shengcheng Yu", + "Zhenyu Chen", + "Chunrong Fang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 25, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "planning-agent" + ], + "arxiv_id": "2606.10749", + "source": "arxiv", + "source_id": "arxiv:2606.10749", + "pdf_url": "https://arxiv.org/pdf/2606.10749", + "primary_query": "agent-safety" + }, + { + "id": "2606.29824", + "title": "Neural Procedural Memory: Empowering LLM Agents with Implicit Activation Steering", + "url": "https://arxiv.org/abs/2606.29824", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Chengfeng Zhao", + "Yuqiao Tan", + "Shizhu He", + "Yequan Wang", + "Jun Zhao", + "Kang Liu" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "agent-memory", + "autonomous-agent-llm", + "llm-agent", + "rag-agent" + ], + "arxiv_id": "2606.29824", + "source": "arxiv", + "source_id": "arxiv:2606.29824", + "pdf_url": "https://arxiv.org/pdf/2606.29824", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.28011", + "title": "From Detection to Action: Using LLM Agents for Fault-Tolerant Control", + "url": "https://arxiv.org/abs/2606.28011", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Javal Vyas", + "Milapji Singh Gill", + "Artan Markaj", + "Felix Gehlhoff", + "Mehmet Mercangöz" + ], + "categories": [ + "eess.SY", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent", + "multi-agent-llm", + "planning-agent", + "rag-agent" + ], + "arxiv_id": "2606.28011", + "source": "arxiv", + "source_id": "arxiv:2606.28011", + "pdf_url": "https://arxiv.org/pdf/2606.28011", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20401", + "title": "PowerAgentBench-Dyn: A Benchmark for Agentic AI in Power System Dynamic Studies", + "url": "https://arxiv.org/abs/2606.20401", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Qian Zhang", + "Andrea Pomarico", + "Costas Mylonas", + "Magda Foti", + "Alberto Berizzi", + "Le Xie" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.20401", + "source": "arxiv", + "source_id": "arxiv:2606.20401", + "pdf_url": "https://arxiv.org/pdf/2606.20401", + "primary_query": "agentic-ai" + }, + { + "id": "2606.18789", + "title": "PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies", + "url": "https://arxiv.org/abs/2606.18789", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Costas Mylonas", + "Magda Foti", + "Andrea Pomarico", + "Matheus Duarte", + "Qian Zhang", + "Emmanouel Varvarigos" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2606.18789", + "source": "arxiv", + "source_id": "arxiv:2606.18789", + "pdf_url": "https://arxiv.org/pdf/2606.18789", + "primary_query": "agentic-ai" + }, + { + "id": "2606.08274", + "title": "Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents", + "url": "https://arxiv.org/abs/2606.08274", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Safia Baloch", + "Rahemeen Khan" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 24, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2606.08274", + "source": "arxiv", + "source_id": "arxiv:2606.08274", + "pdf_url": "https://arxiv.org/pdf/2606.08274", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.05775", + "title": "Beyond the Leaderboard: A Synthesis of Tool-Use, Planning, and Reasoning Failures in Large Language Model Agents", + "url": "https://arxiv.org/abs/2607.05775", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Wael Albayaydh", + "Rui Zhao", + "Ivan Flechais" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "embodied-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2607.05775", + "source": "arxiv", + "source_id": "arxiv:2607.05775", + "pdf_url": "https://arxiv.org/pdf/2607.05775", + "primary_query": "llm-agent" + }, + { + "id": "2607.02255", + "title": "AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2607.02255", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Xiangchen Cheng", + "Yunwei Jiang", + "Jianwen Sun", + "Zizhen Li", + "Chuanhao Li", + "Xiangcheng Cao", + "Yihao Liu", + "Fanrui Zhang", + "Li Jin", + "Kaipeng Zhang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.02255", + "source": "arxiv", + "source_id": "arxiv:2607.02255", + "pdf_url": "https://arxiv.org/pdf/2607.02255", + "primary_query": "llm-agent" + }, + { + "id": "2606.28791", + "title": "From Determinism to Delegation: AI-Native Software Engineering and the Evolution of the Agentic Engineer", + "url": "https://arxiv.org/abs/2606.28791", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Mamdouh Alenezi" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "autonomous-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.28791", + "source": "arxiv", + "source_id": "arxiv:2606.28791", + "pdf_url": "https://arxiv.org/pdf/2606.28791", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26614", + "title": "HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization", + "url": "https://arxiv.org/abs/2606.26614", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Kuangshi Ai", + "Patrick Phuoc Do", + "Chaoli Wang" + ], + "categories": [ + "cs.HC", + "cs.AI", + "cs.GR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "planning-agent" + ], + "arxiv_id": "2606.26614", + "source": "arxiv", + "source_id": "arxiv:2606.26614", + "pdf_url": "https://arxiv.org/pdf/2606.26614", + "primary_query": "multi-agent-llm" + }, + { + "id": "2605.08442", + "title": "Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents", + "url": "https://arxiv.org/abs/2605.08442", + "published": "2026-05-08", + "updated": "2026-07-03", + "authors": [ + "Jun Wen Leong" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.08442", + "source": "arxiv", + "source_id": "arxiv:2605.08442", + "pdf_url": "https://arxiv.org/pdf/2605.08442", + "primary_query": "agent-safety" + }, + { + "id": "2605.06869", + "title": "Agentick: A Unified Benchmark for General Sequential Decision-Making Agents", + "url": "https://arxiv.org/abs/2605.06869", + "published": "2026-05-07", + "updated": "2026-05-12", + "authors": [ + "Roger Creus Castanyer", + "Pablo Samuel Castro", + "Glen Berseth" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 23, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.06869", + "source": "arxiv", + "source_id": "arxiv:2605.06869", + "pdf_url": "https://arxiv.org/pdf/2605.06869", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.03233", + "title": "Agentic and Generative AI for Open-Source Intelligence and Cyber Investigations: Taxonomy, Evaluation, Challenges, and Future Directions", + "url": "https://arxiv.org/abs/2607.03233", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Eduardo Almeida Palmieri", + "Mohamed Chahine Ghanem", + "Dipo Dunsin", + "Zubair Baig", + "Ed de Quincey", + "Kim-Kwang Raymond Choo" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.IR", + "cs.SI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "rag-agent", + "tool-use" + ], + "arxiv_id": "2607.03233", + "source": "arxiv", + "source_id": "arxiv:2607.03233", + "pdf_url": "https://arxiv.org/pdf/2607.03233", + "primary_query": "agentic-ai" + }, + { + "id": "2606.28061", + "title": "ToolPrivacyBench: Benchmarking Purpose-Bound Privacy in Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.28061", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Shijing Hu", + "Liang Liu", + "Zhu Meng", + "Zhicheng Zhao" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "function-calling", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.28061", + "source": "arxiv", + "source_id": "arxiv:2606.28061", + "pdf_url": "https://arxiv.org/pdf/2606.28061", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.17459", + "title": "Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation", + "url": "https://arxiv.org/abs/2606.17459", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Yuyang Dai", + "Xueqing Peng", + "Lingfei Qian", + "Zhuohan Xie" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "planning-agent" + ], + "arxiv_id": "2606.17459", + "source": "arxiv", + "source_id": "arxiv:2606.17459", + "pdf_url": "https://arxiv.org/pdf/2606.17459", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.16613", + "title": "CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies", + "url": "https://arxiv.org/abs/2606.16613", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Issa Sugiura", + "Daichi Hattori", + "Kazuo Araragi", + "Keita Ogawa", + "Shota Onose", + "Taro Makino", + "Teppei Usuki", + "Takashi Ishida" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2606.16613", + "source": "arxiv", + "source_id": "arxiv:2606.16613", + "pdf_url": "https://arxiv.org/pdf/2606.16613", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.12945", + "title": "Learning What to Remember: A Cognitively Grounded Multi-Factor Value Model for Agentic Memory", + "url": "https://arxiv.org/abs/2606.12945", + "published": "2026-06-11", + "updated": "2026-06-20", + "authors": [ + "Zhibao Chen", + "Qian Cheng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.12945", + "source": "arxiv", + "source_id": "arxiv:2606.12945", + "pdf_url": "https://arxiv.org/pdf/2606.12945", + "primary_query": "agent-memory" + }, + { + "id": "2606.06399", + "title": "CollabSim: A CSCW-Grounded Methodology for Investigating Collaborative Competence of LLM Agents through Controlled Multi-Agent Experiments", + "url": "https://arxiv.org/abs/2606.06399", + "published": "2026-06-04", + "updated": "2026-06-06", + "authors": [ + "Jiaju Chen", + "Bo Sun", + "Yuxuan Lu", + "Yun Wang", + "Dakuo Wang", + "Bingsheng Yao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.06399", + "source": "arxiv", + "source_id": "arxiv:2606.06399", + "pdf_url": "https://arxiv.org/pdf/2606.06399", + "primary_query": "planning-agent" + }, + { + "id": "2606.01199", + "title": "Can LLM Agents Sustain Long-Horizon Organizational Dynamics?", + "url": "https://arxiv.org/abs/2606.01199", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Xuancheng Zhu", + "Yang Yue", + "Shuaibing Wan", + "Zihan Dou", + "Xiaohan Zhang", + "Yongrui Liu", + "Guoshun Nan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "workflow-agent", + "world-model" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "language-agent", + "planning-agent" + ], + "arxiv_id": "2606.01199", + "source": "arxiv", + "source_id": "arxiv:2606.01199", + "pdf_url": "https://arxiv.org/pdf/2606.01199", + "primary_query": "language-agent" + }, + { + "id": "2605.29861", + "title": "Towards Verifiable Multimodal Deep Research: A Multi-Agent Harness for Interleaved Report Generation", + "url": "https://arxiv.org/abs/2605.29861", + "published": "2026-05-28", + "updated": "2026-06-03", + "authors": [ + "Chenghao Zhang", + "Guanting Dong", + "Yufan Liu", + "Tong Zhao", + "Xiaoxi Li", + "Zhicheng Dou" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.29861", + "source": "arxiv", + "source_id": "arxiv:2605.29861", + "pdf_url": "https://arxiv.org/pdf/2605.29861", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.18652", + "title": "MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents", + "url": "https://arxiv.org/abs/2605.18652", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Ziyun Zeng", + "Hang Hua", + "Bocheng Zou", + "Mu Cai", + "Rogerio Feris", + "Jiebo Luo" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.18652", + "source": "arxiv", + "source_id": "arxiv:2605.18652", + "pdf_url": "https://arxiv.org/pdf/2605.18652", + "primary_query": "agent-memory" + }, + { + "id": "2605.05704", + "title": "SafeHarbor: Hierarchical Memory-Augmented Guardrail for LLM Agent Safety", + "url": "https://arxiv.org/abs/2605.05704", + "published": "2026-05-07", + "updated": "2026-05-22", + "authors": [ + "Zhe Liu", + "Zonghao Ying", + "Wenxin Zhang", + "Quanchen Zou", + "Deyue Zhang", + "Dongdong Yang", + "Xiangzheng Zhang", + "Hao Peng" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "computer-use", + "memory", + "reasoning", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "autonomous-agent-llm" + ], + "arxiv_id": "2605.05704", + "source": "arxiv", + "source_id": "arxiv:2605.05704", + "pdf_url": "https://arxiv.org/pdf/2605.05704", + "primary_query": "agent-safety" + }, + { + "id": "2605.03242", + "title": "Enhancing Agent Safety Judgment: Controlled Benchmark Rewriting and Analogical Reasoning for Deceptive Out-of-Distribution Scenarios", + "url": "https://arxiv.org/abs/2605.03242", + "published": "2026-05-05", + "updated": "2026-05-05", + "authors": [ + "Zuoyu Zhang", + "Yancheng Zhu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 22, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.03242", + "source": "arxiv", + "source_id": "arxiv:2605.03242", + "pdf_url": "https://arxiv.org/pdf/2605.03242", + "primary_query": "agent-safety" + }, + { + "id": "2607.06118", + "title": "WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation", + "url": "https://arxiv.org/abs/2607.06118", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Wei Dong", + "Tianyu Fu", + "Zhe Yu", + "Hanning Wang", + "Anyang Su", + "Zhizhou Fang", + "Yuyang Chen", + "Shuo Wang", + "Minghui Wu", + "Ping Jiang", + "Zhen Lei", + "Chenxu Zhao" + ], + "categories": [ + "cs.CV", + "cs.MM" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "web-gui-agent" + ], + "arxiv_id": "2607.06118", + "source": "arxiv", + "source_id": "arxiv:2607.06118", + "pdf_url": "https://arxiv.org/pdf/2607.06118", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.05773", + "title": "Beyond Static Evaluation: Building Simulation Environments for Scalable Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2607.05773", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Akshay Arora", + "Ishan Nigam", + "Ashutosh Aggarwal", + "Shefali Bansal", + "Krishna Singh", + "Sweta Kumari", + "Nikhil Mittal", + "Shariq Farhan", + "Siddarth Malreddy" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "world-model" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "tool-use", + "web-gui-agent" + ], + "arxiv_id": "2607.05773", + "source": "arxiv", + "source_id": "arxiv:2607.05773", + "pdf_url": "https://arxiv.org/pdf/2607.05773", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.05456", + "title": "Prompt-to-Paper: Agentic AI System for Bioinformatics", + "url": "https://arxiv.org/abs/2607.05456", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Ramsha Kamran", + "Maheera Amjad", + "Zartasha Mustansar", + "Arsalan Shaukat", + "Salma Sherbaz", + "Muhammad U. S. Khan" + ], + "categories": [ + "cs.AI", + "cs.CL", + "q-bio.QM" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.05456", + "source": "arxiv", + "source_id": "arxiv:2607.05456", + "pdf_url": "https://arxiv.org/pdf/2607.05456", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04433", + "title": "Autonomous Information Seeking: A Roadmap for Agentic Recommender Systems", + "url": "https://arxiv.org/abs/2607.04433", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Xinyu Lin", + "Yashar Deldjoo", + "Sunhao Dai", + "Honghui Bao", + "Xiaopeng Ye", + "Fatemeh Nazary", + "Wenjie Wang", + "Tommaso Di Noia", + "Jun Xu", + "Tat-Seng Chua" + ], + "categories": [ + "cs.IR", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.04433", + "source": "arxiv", + "source_id": "arxiv:2607.04433", + "pdf_url": "https://arxiv.org/pdf/2607.04433", + "primary_query": "tool-use" + }, + { + "id": "2607.03601", + "title": "ArchEval: Measuring AI Agents as Computer Architects", + "url": "https://arxiv.org/abs/2607.03601", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Chenyu Wang", + "Zishen Wan", + "Jeffrey Ma", + "Shvetank Prakash", + "Zhenting Qi", + "Haebin Do", + "Andy Cheng", + "Arya Tschand", + "Jiahe Shi", + "Yilun Du", + "Vijay Janapa Reddi" + ], + "categories": [ + "cs.AR" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.03601", + "source": "arxiv", + "source_id": "arxiv:2607.03601", + "pdf_url": "https://arxiv.org/pdf/2607.03601", + "primary_query": "ai-agent" + }, + { + "id": "2607.02032", + "title": "PACE: A Proxy for Agentic Capability Evaluation", + "url": "https://arxiv.org/abs/2607.02032", + "published": "2026-07-02", + "updated": "2026-07-06", + "authors": [ + "Yueqi Song", + "Lintang Sutawika", + "Jiarui Liu", + "Lindia Tjuatja", + "Jiayi Geng", + "Yunze Xiao", + "Daniel Lee", + "Aditya Bharat Soni", + "Vincent Lo", + "Xiang Yue", + "Graham Neubig" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.02032", + "source": "arxiv", + "source_id": "arxiv:2607.02032", + "pdf_url": "https://arxiv.org/pdf/2607.02032", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.01641", + "title": "When Agents Do Not Stop: Uncovering Infinite Agentic Loops in LLM Agents", + "url": "https://arxiv.org/abs/2607.01641", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Xinyi Hou", + "Shenao Wang", + "Yanjie Zhao", + "Haoyu Wang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2607.01641", + "source": "arxiv", + "source_id": "arxiv:2607.01641", + "pdf_url": "https://arxiv.org/pdf/2607.01641", + "primary_query": "llm-agent" + }, + { + "id": "2606.30524", + "title": "The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems", + "url": "https://arxiv.org/abs/2606.30524", + "published": "2026-06-29", + "updated": "2026-07-01", + "authors": [ + "Abu Saleh", + "Tesfay Welegebreal Tesfay", + "Phuong T. Nguyen", + "Juri Di Rocco", + "Muhammad Umar Zeshan", + "Davide Di Ruscio" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "planning", + "rag" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "rag-agent" + ], + "arxiv_id": "2606.30524", + "source": "arxiv", + "source_id": "arxiv:2606.30524", + "pdf_url": "https://arxiv.org/pdf/2606.30524", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29537", + "title": "OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks", + "url": "https://arxiv.org/abs/2606.29537", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Mengqi Yuan", + "Zilong Zhou", + "Xinzhuang Xiong", + "Weiming Wu", + "Jiayang Sun", + "Jiamin Song", + "Kaiqian Cui", + "Bowen Wang", + "Haoyuan Wu", + "Yitong Li", + "Dunjie Lu", + "Haikong Lu", + "Qi Zhen", + "Xinyuan Wang", + "Jiaqi Deng", + "Yuhao Yang", + "Cheng Chen", + "Boyuan Zheng", + "Alex Su", + "Xiao Yu", + "Hao Zou", + "Saaket Agashe", + "Xing Han Lu", + "Manpreet Kaur", + "Zhengyang Qi", + "Vincent Sunn Chen", + "Frederic Sala", + "Dayiheng Liu", + "Junyang Lin", + "Zhou Yu", + "Yu Su", + "Siva Reddy", + "Xin Eric Wang", + "Peng Qi", + "Tianbao Xie", + "Tao Yu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.29537", + "source": "arxiv", + "source_id": "arxiv:2606.29537", + "pdf_url": "https://arxiv.org/pdf/2606.29537", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.28925", + "title": "Multi-Agent Routing as Set-Valued Prediction: A WildChat Benchmark and Cost-Aware Evaluation", + "url": "https://arxiv.org/abs/2606.28925", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Ananto Nayan Bala", + "Faisal Muhammad Shah" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.IR", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28925", + "source": "arxiv", + "source_id": "arxiv:2606.28925", + "pdf_url": "https://arxiv.org/pdf/2606.28925", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.26511", + "title": "Temporal Validity in Retrieval Memory: Eliminating Stale-Fact Errors for AI Agents over Evolving Knowledge", + "url": "https://arxiv.org/abs/2606.26511", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Neeraj Yadav" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.ET", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2606.26511", + "source": "arxiv", + "source_id": "arxiv:2606.26511", + "pdf_url": "https://arxiv.org/pdf/2606.26511", + "primary_query": "ai-agent" + }, + { + "id": "2606.22557", + "title": "MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop", + "url": "https://arxiv.org/abs/2606.22557", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Yikun Fu", + "Bowen Fu", + "Zhenyu Wu", + "Shuang Cheng", + "Xiaowei Sun", + "Bowen Yang", + "Zehao Li", + "Yibo Zhao", + "Zichen Ding", + "Zhoumianze Liu", + "Shijie Wang", + "Biqing Qi", + "Bowen Zhou" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "ai-agent", + "web-gui-agent" + ], + "arxiv_id": "2606.22557", + "source": "arxiv", + "source_id": "arxiv:2606.22557", + "pdf_url": "https://arxiv.org/pdf/2606.22557", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.21877", + "title": "AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems", + "url": "https://arxiv.org/abs/2606.21877", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Srimonti Dutta", + "Akshata Kishore Moharir" + ], + "categories": [ + "cs.AI", + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "rag-agent", + "tool-use" + ], + "arxiv_id": "2606.21877", + "source": "arxiv", + "source_id": "arxiv:2606.21877", + "pdf_url": "https://arxiv.org/pdf/2606.21877", + "primary_query": "agentic-ai" + }, + { + "id": "2606.16802", + "title": "LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control", + "url": "https://arxiv.org/abs/2606.16802", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Anqi Zou", + "Han Deng", + "Chengyu Zhang", + "Junquan Hu", + "Yu Wang", + "Yuxiang Xing", + "Aokai Zhang", + "Hanling Zhang", + "Zhaoyang Liu", + "Ben Fei", + "Zhihui Wang", + "Wanli Ouyang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.16802", + "source": "arxiv", + "source_id": "arxiv:2606.16802", + "pdf_url": "https://arxiv.org/pdf/2606.16802", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13994", + "title": "Hidden in Plain Sight: Benchmarking Agent Safety Against Decomposition Attacks with DECOMPBENCH", + "url": "https://arxiv.org/abs/2606.13994", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Vikhyath Kothamasu", + "Virginia Smith", + "Chhavi Yadav" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "tool-use" + ], + "arxiv_id": "2606.13994", + "source": "arxiv", + "source_id": "arxiv:2606.13994", + "pdf_url": "https://arxiv.org/pdf/2606.13994", + "primary_query": "agent-safety" + }, + { + "id": "2606.04990", + "title": "From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents", + "url": "https://arxiv.org/abs/2606.04990", + "published": "2026-06-03", + "updated": "2026-06-28", + "authors": [ + "Yiqi Wang", + "Jiaqi Zhang", + "Taotao Cai", + "Zirui Liu", + "Qingqiang Sun", + "Zequn Sun", + "Zhangkai Wu", + "Manqing Dong", + "Mingkai Zheng", + "Xuefei Yin", + "Yanming Zhu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.04990", + "source": "arxiv", + "source_id": "arxiv:2606.04990", + "pdf_url": "https://arxiv.org/pdf/2606.04990", + "primary_query": "planning-agent" + }, + { + "id": "2606.02461", + "title": "AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents", + "url": "https://arxiv.org/abs/2606.02461", + "published": "2026-06-01", + "updated": "2026-06-02", + "authors": [ + "Yiheng Shu", + "Bernal Jiménez Gutiérrez", + "Saisri Padmaja Jonnalagedda", + "Yuguang Yao", + "Huan Sun", + "Yu Su" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.02461", + "source": "arxiv", + "source_id": "arxiv:2606.02461", + "pdf_url": "https://arxiv.org/pdf/2606.02461", + "primary_query": "language-agent" + }, + { + "id": "2606.01385", + "title": "Bridging Requirements and Architecture: Multi-Agent Orchestration with External Knowledge and Hierarchical Memory", + "url": "https://arxiv.org/abs/2606.01385", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Ruiyin Li", + "Yiran Zhang", + "Xiyu Zhou", + "Yangxiao Cai", + "Peng Liang", + "Weisong Sun", + "Jifeng Xuan", + "Zhi Jin", + "Yang Liu" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "multi-agent", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.01385", + "source": "arxiv", + "source_id": "arxiv:2606.01385", + "pdf_url": "https://arxiv.org/pdf/2606.01385", + "primary_query": "rag-agent" + }, + { + "id": "2606.00756", + "title": "CoMIC: Collaborative Memory and Insights Circulation for Long-Horizon LLM Agents in Cloud-Edge Systems", + "url": "https://arxiv.org/abs/2606.00756", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Yannan Wang", + "Longli Yang", + "Zhen Liu", + "Abhishek Kumar", + "Carsten Maple" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.00756", + "source": "arxiv", + "source_id": "arxiv:2606.00756", + "pdf_url": "https://arxiv.org/pdf/2606.00756", + "primary_query": "planning-agent" + }, + { + "id": "2605.20315", + "title": "Mix-Quant: Quantized Prefilling, Precise Decoding for Agentic LLMs", + "url": "https://arxiv.org/abs/2605.20315", + "published": "2026-05-19", + "updated": "2026-05-19", + "authors": [ + "Haiquan Lu", + "Zigeng Chen", + "Gongfan Fang", + "Xinyin Ma", + "Xinchao Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.20315", + "source": "arxiv", + "source_id": "arxiv:2605.20315", + "pdf_url": "https://arxiv.org/pdf/2605.20315", + "primary_query": "planning-agent" + }, + { + "id": "2605.14498", + "title": "GroupMemBench: Benchmarking LLM Agent Memory in Multi-Party Conversations", + "url": "https://arxiv.org/abs/2605.14498", + "published": "2026-05-14", + "updated": "2026-05-16", + "authors": [ + "Jingbo Yang", + "Kwei-Herng Lai", + "Xiaowen Wang", + "Shiyu Chang", + "Yaar Harari", + "Evgeniy Gabrilovich" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.14498", + "source": "arxiv", + "source_id": "arxiv:2605.14498", + "pdf_url": "https://arxiv.org/pdf/2605.14498", + "primary_query": "agent-memory" + }, + { + "id": "2605.11633", + "title": "Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations", + "url": "https://arxiv.org/abs/2605.11633", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Junjue Wang", + "Weihao Xuan", + "Heli Qi", + "Pengyu Dai", + "Kunyi Liu", + "Hongruixuan Chen", + "Zhuo Zheng", + "Junshi Xia", + "Stefano Ermon", + "Naoto Yokoya" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.11633", + "source": "arxiv", + "source_id": "arxiv:2605.11633", + "pdf_url": "https://arxiv.org/pdf/2605.11633", + "primary_query": "planning-agent" + }, + { + "id": "2605.06812", + "title": "Towards Security-Auditable LLM Agents: A Unified Graph Representation", + "url": "https://arxiv.org/abs/2605.06812", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Chaofan Li", + "Lyuye Zhang", + "Jintao Zhai", + "Siyue Feng", + "Xichun Yang", + "Huahao Wang", + "Shihan Dou", + "Yu Ji", + "Yutao Hu", + "Yueming Wu", + "Yang Liu", + "Deqing Zou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.06812", + "source": "arxiv", + "source_id": "arxiv:2605.06812", + "pdf_url": "https://arxiv.org/pdf/2605.06812", + "primary_query": "agent-safety" + }, + { + "id": "2602.08412", + "title": "From Assistant to Double Agent: Formalizing and Benchmarking Attacks on OpenClaw for Personalized Local AI Agent", + "url": "https://arxiv.org/abs/2602.08412", + "published": "2026-02-09", + "updated": "2026-02-11", + "authors": [ + "Yuhang Wang", + "Feiming Xu", + "Zheng Lin", + "Guangyu He", + "Yuzhe Huang", + "Haichang Gao", + "Zhenxing Niu", + "Shiguo Lian", + "Zhaoxiang Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.08412", + "source": "arxiv", + "source_id": "arxiv:2602.08412", + "pdf_url": "https://arxiv.org/pdf/2602.08412", + "primary_query": "agent-safety" + }, + { + "id": "2508.07575", + "title": "MCPToolBench++: A Large Scale AI Agent Model Context Protocol MCP Tool Use Benchmark", + "url": "https://arxiv.org/abs/2508.07575", + "published": "2025-08-11", + "updated": "2025-08-11", + "authors": [ + "Shiqing Fan", + "Xichen Ding", + "Liang Zhang", + "Linjian Mo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 21, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.07575", + "source": "arxiv", + "source_id": "arxiv:2508.07575", + "pdf_url": "https://arxiv.org/pdf/2508.07575", + "primary_query": "function-calling" + }, + { + "id": "2607.05318", + "title": "PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems", + "url": "https://arxiv.org/abs/2607.05318", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Shubham Gupta", + "Nazanin Mohammadi Sepahvand", + "Abhinav Kumar", + "Cem Subakan", + "Spandana Gella", + "Pierre-André Noël", + "Perouz Taslakian", + "Eugene Bagdasarian", + "Valentina Zantedeschi" + ], + "categories": [ + "cs.MA", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.05318", + "source": "arxiv", + "source_id": "arxiv:2607.05318", + "pdf_url": "https://arxiv.org/pdf/2607.05318", + "primary_query": "llm-agent" + }, + { + "id": "2607.04391", + "title": "Memory-Orchestrated Semantic System (MOSS): An Auditable Agentic Memory Architecture", + "url": "https://arxiv.org/abs/2607.04391", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Serge Lacasse", + "Jérémie Hatier", + "Alex Baker" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2607.04391", + "source": "arxiv", + "source_id": "arxiv:2607.04391", + "pdf_url": "https://arxiv.org/pdf/2607.04391", + "primary_query": "agent-memory" + }, + { + "id": "2607.03953", + "title": "The Remarkable Effectiveness of Providing AI Agents with Natural Language Tools: A Replication Study Validating NLT Performance Across 14 Models", + "url": "https://arxiv.org/abs/2607.03953", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Alexander Somma", + "Isabelle Plante", + "Fred Premji" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.03953", + "source": "arxiv", + "source_id": "arxiv:2607.03953", + "pdf_url": "https://arxiv.org/pdf/2607.03953", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03726", + "title": "SelfMem: Self-Optimizing Memory for AI Agents", + "url": "https://arxiv.org/abs/2607.03726", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Shu Yang", + "Junchao Wu", + "Derek F. Wong", + "Di Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "ai-agent", + "tool-use" + ], + "arxiv_id": "2607.03726", + "source": "arxiv", + "source_id": "arxiv:2607.03726", + "pdf_url": "https://arxiv.org/pdf/2607.03726", + "primary_query": "agent-memory" + }, + { + "id": "2607.01766", + "title": "SimWorlds: A Multi-Agent System for Dynamic 3D Scene Creation", + "url": "https://arxiv.org/abs/2607.01766", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Chunjiang Liu", + "Xiaoyuan Wang", + "Haoyu Chen", + "Yizhou Zhao", + "Ming-Hsuan Yang", + "László A. Jeni" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.01766", + "source": "arxiv", + "source_id": "arxiv:2607.01766", + "pdf_url": "https://arxiv.org/pdf/2607.01766", + "primary_query": "llm-agent" + }, + { + "id": "2607.00454", + "title": "Agri-SAGE: Simulation-Grounded Multi-Agent LLM for Context-Aware Agricultural Advisory Generation", + "url": "https://arxiv.org/abs/2607.00454", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Vedant Balasubramaniam", + "Geetha Charan", + "Manojkumar Patil", + "Rohit P Suresh", + "V Priyanka", + "Kodur Sai Vinay Sathvik", + "Y. Narahari" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "world-model" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00454", + "source": "arxiv", + "source_id": "arxiv:2607.00454", + "pdf_url": "https://arxiv.org/pdf/2607.00454", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31073", + "title": "MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning", + "url": "https://arxiv.org/abs/2606.31073", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Sheng Zhang", + "Qinglin Li", + "Yuechao Zang", + "Xueqin Huang", + "Yijia Fu", + "Cheng Zhu" + ], + "categories": [ + "cs.AI", + "cs.MA", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use", + "world-model" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.31073", + "source": "arxiv", + "source_id": "arxiv:2606.31073", + "pdf_url": "https://arxiv.org/pdf/2606.31073", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.31179", + "title": "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents", + "url": "https://arxiv.org/abs/2606.31179", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Qianchu Liu", + "Sheng Zhang", + "Guanghui Qin", + "Jeya Maria Jose Valanarasu", + "Maximilian Rokuss", + "Mingyu Lu", + "Timothy Ossowski", + "Juan Manuel Zambrano Chaves", + "Cliff Wong", + "Peniel Argaw", + "Yashna Hasija", + "Mu Wei", + "Wen-wai Yim", + "Qin Liu", + "Zilin Jing", + "Jason Entenmann", + "Naoto Usuyama", + "Tristan Naumann", + "Hoifung Poon" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "ai-agent" + ], + "arxiv_id": "2606.31179", + "source": "arxiv", + "source_id": "arxiv:2606.31179", + "pdf_url": "https://arxiv.org/pdf/2606.31179", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.31612", + "title": "What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States", + "url": "https://arxiv.org/abs/2606.31612", + "published": "2026-06-30", + "updated": "2026-07-02", + "authors": [ + "Chen Liu", + "Ling Chen", + "Hanzhang Zhou", + "Xu Zhang", + "Quyu Kong", + "Panrong Tong", + "Wenhao Wang", + "Xin Yu", + "Steven Hoi", + "Yue Wang" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "web-gui-agent" + ], + "arxiv_id": "2606.31612", + "source": "arxiv", + "source_id": "arxiv:2606.31612", + "pdf_url": "https://arxiv.org/pdf/2606.31612", + "primary_query": "agent-memory" + }, + { + "id": "2606.30906", + "title": "Investigating Multi-Agent Deliberation in Law", + "url": "https://arxiv.org/abs/2606.30906", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Cor Steging", + "Ludi van Leeuwen", + "Tadeusz Zbiegień" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.30906", + "source": "arxiv", + "source_id": "arxiv:2606.30906", + "pdf_url": "https://arxiv.org/pdf/2606.30906", + "primary_query": "agentic-ai" + }, + { + "id": "2606.30949", + "title": "AgRefactor: Self-Evolving Agentic Workflow for HLS Compatibility and Performance", + "url": "https://arxiv.org/abs/2606.30949", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yang Zou", + "Zijian Ding", + "Yizhou Sun", + "Jason Cong" + ], + "categories": [ + "cs.AI", + "cs.AR" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.30949", + "source": "arxiv", + "source_id": "arxiv:2606.30949", + "pdf_url": "https://arxiv.org/pdf/2606.30949", + "primary_query": "agentic-ai" + }, + { + "id": "2606.29116", + "title": "Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem", + "url": "https://arxiv.org/abs/2606.29116", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Yutian Tang", + "Yuming Zhou", + "Huaming Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent", + "planning-agent", + "tool-use" + ], + "arxiv_id": "2606.29116", + "source": "arxiv", + "source_id": "arxiv:2606.29116", + "pdf_url": "https://arxiv.org/pdf/2606.29116", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24535", + "title": "Governed Shared Memory for Multi-Agent LLM Systems", + "url": "https://arxiv.org/abs/2606.24535", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yanki Margalit", + "Nurit Cohen-Inger", + "Erni Avram", + "Ran Taig", + "Oded Margalit" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "multi-agent-llm" + ], + "arxiv_id": "2606.24535", + "source": "arxiv", + "source_id": "arxiv:2606.24535", + "pdf_url": "https://arxiv.org/pdf/2606.24535", + "primary_query": "agent-memory" + }, + { + "id": "2606.24820", + "title": "SHERLOC: Structured Diagnostic Localization for Code Repair Agents", + "url": "https://arxiv.org/abs/2606.24820", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Hovhannes Tamoyan", + "Sean Narenthiran", + "Erik Arakelyan", + "Mira Mezini", + "Boris Ginsburg" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "multi-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.24820", + "source": "arxiv", + "source_id": "arxiv:2606.24820", + "pdf_url": "https://arxiv.org/pdf/2606.24820", + "primary_query": "coding-agent" + }, + { + "id": "2606.22263", + "title": "Revelio: Cost-Efficient Agentic Memory Safety Vulnerability Detection For Repository-Scale Codebases", + "url": "https://arxiv.org/abs/2606.22263", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Yiwei Hou", + "Hao Wang", + "Muxi Lyu", + "Marius Momeu", + "Eric Nguyen", + "Taige Yang", + "Koushik Sen", + "Dawn Song", + "David Wagner" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.MA", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "coding-agent" + ], + "arxiv_id": "2606.22263", + "source": "arxiv", + "source_id": "arxiv:2606.22263", + "pdf_url": "https://arxiv.org/pdf/2606.22263", + "primary_query": "agent-memory" + }, + { + "id": "2606.21627", + "title": "Counsel: A Meta-Evaluation Dataset for Agentic Tasks", + "url": "https://arxiv.org/abs/2606.21627", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Sashank Pisupati", + "Henry Broomfield", + "Eujeong Choi", + "Antonia Calvi", + "Charlie Wang", + "Roman Engeler", + "Max Bartolo", + "Patrick Lewis" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2606.21627", + "source": "arxiv", + "source_id": "arxiv:2606.21627", + "pdf_url": "https://arxiv.org/pdf/2606.21627", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.16871", + "title": "Human-on-the-Bridge: Scalable Evaluation for AI Agents", + "url": "https://arxiv.org/abs/2606.16871", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Fouad Bousetouane" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.16871", + "source": "arxiv", + "source_id": "arxiv:2606.16871", + "pdf_url": "https://arxiv.org/pdf/2606.16871", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.17246", + "title": "GeoDisaster: Benchmarking Orchestrated Agents for Operational Disaster Geo-Intelligence", + "url": "https://arxiv.org/abs/2606.17246", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Maram Hasan", + "Aman Verma", + "Savitra Roy", + "Hariseetharam Gunduboina", + "Daksh Jain", + "Muhammad Haris Khan", + "Subhasis Chaudhuri", + "Biplab Banerjee" + ], + "categories": [ + "cs.CV", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.17246", + "source": "arxiv", + "source_id": "arxiv:2606.17246", + "pdf_url": "https://arxiv.org/pdf/2606.17246", + "primary_query": "tool-use" + }, + { + "id": "2606.17114", + "title": "An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios", + "url": "https://arxiv.org/abs/2606.17114", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Hankyul Baek", + "Jaewon Noh", + "Sang Seo", + "Yongsu Kim", + "Gabriel Waikin Loh Matienzo", + "Young Il Kim", + "Ee Wei Seah", + "Akriti Vij" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "tool-use" + ], + "arxiv_id": "2606.17114", + "source": "arxiv", + "source_id": "arxiv:2606.17114", + "pdf_url": "https://arxiv.org/pdf/2606.17114", + "primary_query": "agent-safety" + }, + { + "id": "2606.16420", + "title": "Transferable Self-Evolving Playbooks for Agentic Security Auditing", + "url": "https://arxiv.org/abs/2606.16420", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Ziyue Wang", + "Cheuk Wang Maurice Ng", + "Chenchen Yu", + "Strick Sheng", + "Kaihua Qin", + "Liyi Zhou" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "tool-use" + ], + "arxiv_id": "2606.16420", + "source": "arxiv", + "source_id": "arxiv:2606.16420", + "pdf_url": "https://arxiv.org/pdf/2606.16420", + "primary_query": "agent-safety" + }, + { + "id": "2606.14790", + "title": "XFlow: An Executable Protocol Programming System for Reliable Multi-Agent Workflows", + "url": "https://arxiv.org/abs/2606.14790", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Hanqi Li", + "Jing Peng", + "Zijian Wang", + "Lu Chen", + "Kai Yu" + ], + "categories": [ + "cs.PL", + "cs.AI" + ], + "topics": [ + "coding-agent", + "memory", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.14790", + "source": "arxiv", + "source_id": "arxiv:2606.14790", + "pdf_url": "https://arxiv.org/pdf/2606.14790", + "primary_query": "tool-use" + }, + { + "id": "2606.08531", + "title": "VESTA: A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents", + "url": "https://arxiv.org/abs/2606.08531", + "published": "2026-06-07", + "updated": "2026-06-07", + "authors": [ + "Lu Jia", + "Haibo Tong", + "Feifei Zhao", + "Jindong Li", + "Dongqi Liang", + "Ping Wu", + "Qian Zhang", + "Yi Zeng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.08531", + "source": "arxiv", + "source_id": "arxiv:2606.08531", + "pdf_url": "https://arxiv.org/pdf/2606.08531", + "primary_query": "agent-safety" + }, + { + "id": "2606.07402", + "title": "M$^3$Exam: Benchmarking Multimodal Memory for Realistic User-Agent Interactions", + "url": "https://arxiv.org/abs/2606.07402", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Zhengjun Huang", + "Wenxuan Liu", + "Zhoujin Tian", + "Wei Chen", + "Junle Chen", + "Yuqian Wu", + "Fangyuan Zhang", + "Qintian Guo", + "Xiaofang Zhou" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.07402", + "source": "arxiv", + "source_id": "arxiv:2606.07402", + "pdf_url": "https://arxiv.org/pdf/2606.07402", + "primary_query": "language-agent" + }, + { + "id": "2606.04780", + "title": "PersonaTree: Structured Lifecycle Memory for Person Understanding in LLM Agents", + "url": "https://arxiv.org/abs/2606.04780", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Yubo Hou", + "Jingwei Song", + "Hongbo Zhang", + "Zhisheng Chen", + "Bang Xiao", + "Tao Wan", + "Zengchang Qin" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04780", + "source": "arxiv", + "source_id": "arxiv:2606.04780", + "pdf_url": "https://arxiv.org/pdf/2606.04780", + "primary_query": "agent-memory" + }, + { + "id": "2606.03657", + "title": "Diagnosing Knowledge Gaps in LLM Tool Use: An Agentic Benchmark for Novel API Acquisition", + "url": "https://arxiv.org/abs/2606.03657", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Jinnuo Liu", + "Yue Peng", + "Jinhan Niu", + "Hongyi Wen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.03657", + "source": "arxiv", + "source_id": "arxiv:2606.03657", + "pdf_url": "https://arxiv.org/pdf/2606.03657", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.02109", + "title": "BADGER: Bridging Agentic and Deterministic Evaluation for Generative Enterprise Reasoning", + "url": "https://arxiv.org/abs/2606.02109", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Shannon Serrao", + "Soumitra Chatterjee", + "Dorina Strori", + "Abhishek Sharma", + "Nathan Miller" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.02109", + "source": "arxiv", + "source_id": "arxiv:2606.02109", + "pdf_url": "https://arxiv.org/pdf/2606.02109", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.01416", + "title": "Self-Healing Agentic Orchestrators for Reliable Tool-Augmented Large Language Model Systems", + "url": "https://arxiv.org/abs/2606.01416", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Rahul Suresh Babu", + "Adarsh Agrawal" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.01416", + "source": "arxiv", + "source_id": "arxiv:2606.01416", + "pdf_url": "https://arxiv.org/pdf/2606.01416", + "primary_query": "planning-agent" + }, + { + "id": "2606.00610", + "title": "MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation", + "url": "https://arxiv.org/abs/2606.00610", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Chuanjie Wu", + "Zhishang Xiang", + "Yunbo Tang", + "Zerui Chen", + "Qinggang Zhang", + "Jinsong Su" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.00610", + "source": "arxiv", + "source_id": "arxiv:2606.00610", + "pdf_url": "https://arxiv.org/pdf/2606.00610", + "primary_query": "rag-agent" + }, + { + "id": "2605.30883", + "title": "TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking", + "url": "https://arxiv.org/abs/2605.30883", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Churui Zeng", + "Weiwei Qi", + "Kedong Xiu", + "Tianhang Zheng", + "Chaochao Lu", + "Liang He", + "Zhan Qin", + "Kui Ren" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.30883", + "source": "arxiv", + "source_id": "arxiv:2605.30883", + "pdf_url": "https://arxiv.org/pdf/2605.30883", + "primary_query": "planning-agent" + }, + { + "id": "2605.30090", + "title": "DirectorBench: Diagnosing Long-Form Video Generation with Personalized Multi-Agent Evaluation", + "url": "https://arxiv.org/abs/2605.30090", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Jiamin Chen", + "Qianben Chen", + "Jiawen Zhang", + "Yidi Wu", + "Yuchen Li", + "Xiaokun Zhang", + "Wangchunshu Zhou", + "Chen Ma" + ], + "categories": [ + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.30090", + "source": "arxiv", + "source_id": "arxiv:2605.30090", + "pdf_url": "https://arxiv.org/pdf/2605.30090", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.27134", + "title": "Scaling, Benchmarking, and Reasoning of Vision-Language Agents for Mobile GUI Navigation", + "url": "https://arxiv.org/abs/2605.27134", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Heng Qu", + "Yike Liu", + "Renren Jin", + "Wenzong Zhang", + "Pengzhi Gao", + "Wei Liu", + "Jian Luan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.27134", + "source": "arxiv", + "source_id": "arxiv:2605.27134", + "pdf_url": "https://arxiv.org/pdf/2605.27134", + "primary_query": "language-agent" + }, + { + "id": "2605.22643", + "title": "Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety", + "url": "https://arxiv.org/abs/2605.22643", + "published": "2026-05-21", + "updated": "2026-05-22", + "authors": [ + "Piercosma Bisconti", + "Matteo Prandi", + "Federico Pierucci", + "Federico Sartore", + "Enrico Panai", + "Laura Caroli", + "Yue Zhu", + "Adam Leon Smith", + "Luca Nannini", + "Marcello Galisai", + "Susanna Cifani", + "Francesco Giarrusso", + "Marcantonio Bracale Syrnikov", + "Daniele Nardi" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.22643", + "source": "arxiv", + "source_id": "arxiv:2605.22643", + "pdf_url": "https://arxiv.org/pdf/2605.22643", + "primary_query": "agent-safety" + }, + { + "id": "2605.15040", + "title": "Orchard: An Open-Source Agentic Modeling Framework", + "url": "https://arxiv.org/abs/2605.15040", + "published": "2026-05-14", + "updated": "2026-05-21", + "authors": [ + "Baolin Peng", + "Wenlin Yao", + "Qianhui Wu", + "Hao Cheng", + "Xiao Yu", + "Rui Yang", + "Tao Ge", + "Alessandro Sordoni", + "Xingdi Yuan", + "Yelong Shen", + "Pengcheng He", + "Tong Zhang", + "Zhou Yu", + "Jianfeng Gao" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.15040", + "source": "arxiv", + "source_id": "arxiv:2605.15040", + "pdf_url": "https://arxiv.org/pdf/2605.15040", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.13542", + "title": "RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation", + "url": "https://arxiv.org/abs/2605.13542", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Chengzhi Shen", + "Weixiang Shen", + "Tobias Susetzky", + "Chen", + "Chen", + "Jun Li", + "Yuyuan Liu", + "Xuepeng Zhang", + "Zhenyu Gong", + "Daniel Rueckert", + "Jiazhen Pan" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.13542", + "source": "arxiv", + "source_id": "arxiv:2605.13542", + "pdf_url": "https://arxiv.org/pdf/2605.13542", + "primary_query": "agent-memory" + }, + { + "id": "2605.12015", + "title": "SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces", + "url": "https://arxiv.org/abs/2605.12015", + "published": "2026-05-12", + "updated": "2026-05-27", + "authors": [ + "Chang Jin", + "An Wang", + "Zeming Wei", + "Kai Wang", + "Biaojie Zeng", + "Qiaosheng Zhang", + "Chao Yang", + "Jingjing Qu", + "Xia Hu", + "Xingcheng Xu" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.12015", + "source": "arxiv", + "source_id": "arxiv:2605.12015", + "pdf_url": "https://arxiv.org/pdf/2605.12015", + "primary_query": "agent-safety" + }, + { + "id": "2605.08374", + "title": "MemQ: Integrating Q-Learning into Self-Evolving Memory Agents over Provenance DAGs", + "url": "https://arxiv.org/abs/2605.08374", + "published": "2026-05-08", + "updated": "2026-05-14", + "authors": [ + "Junwei Liao", + "Haoting Shi", + "Ruiwen Zhou", + "Jiaqian Wang", + "Shengtao Zhang", + "Wei Zhang", + "Ying Wen", + "Zhiyu Li", + "Feiyu Xiong", + "Bo Tang", + "Weinan Zhang", + "Muning Wen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.08374", + "source": "arxiv", + "source_id": "arxiv:2605.08374", + "pdf_url": "https://arxiv.org/pdf/2605.08374", + "primary_query": "function-calling" + }, + { + "id": "2605.15206", + "title": "AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices", + "url": "https://arxiv.org/abs/2605.15206", + "published": "2026-05-01", + "updated": "2026-05-01", + "authors": [ + "Dzung Pham", + "Kleomenis Katevas", + "Ali Shahin Shamsabadi", + "Hamed Haddadi" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.15206", + "source": "arxiv", + "source_id": "arxiv:2605.15206", + "pdf_url": "https://arxiv.org/pdf/2605.15206", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.16282", + "title": "Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents", + "url": "https://arxiv.org/abs/2605.16282", + "published": "2026-04-11", + "updated": "2026-04-11", + "authors": [ + "Miles Q. Li", + "Benjamin C. M. Fung", + "Boyang Li", + "Heba Ismail", + "Farkhund Iqbal" + ], + "categories": [ + "cs.CY", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.16282", + "source": "arxiv", + "source_id": "arxiv:2605.16282", + "pdf_url": "https://arxiv.org/pdf/2605.16282", + "primary_query": "agent-safety" + }, + { + "id": "2604.02022", + "title": "ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis", + "url": "https://arxiv.org/abs/2604.02022", + "published": "2026-04-02", + "updated": "2026-05-13", + "authors": [ + "Yu Li", + "Haoyu Luo", + "Yuejin Xie", + "Yuqian Fu", + "Zhonghao Yang", + "Shuai Shao", + "Qihan Ren", + "Wanying Qu", + "Yanwei Fu", + "Yujiu Yang", + "Jing Shao", + "Xia Hu", + "Dongrui Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.02022", + "source": "arxiv", + "source_id": "arxiv:2604.02022", + "pdf_url": "https://arxiv.org/pdf/2604.02022", + "primary_query": "agent-safety" + }, + { + "id": "2603.16734", + "title": "Differential Harm Propensity in Personalized LLM Agents: The Curious Case of Mental Health Disclosure", + "url": "https://arxiv.org/abs/2603.16734", + "published": "2026-03-17", + "updated": "2026-03-17", + "authors": [ + "Caglar Yildirim" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.16734", + "source": "arxiv", + "source_id": "arxiv:2603.16734", + "pdf_url": "https://arxiv.org/pdf/2603.16734", + "primary_query": "agent-safety" + }, + { + "id": "2509.20998", + "title": "CORE: Full-Path Evaluation of LLM Agents Beyond Final State", + "url": "https://arxiv.org/abs/2509.20998", + "published": "2025-09-25", + "updated": "2025-09-25", + "authors": [ + "Panagiotis Michelakis", + "Yiannis Hadjiyiannis", + "Dimitrios Stamoulis" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "world-model" + ], + "score": 20, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.20998", + "source": "arxiv", + "source_id": "arxiv:2509.20998", + "pdf_url": "https://arxiv.org/pdf/2509.20998", + "primary_query": "function-calling" + }, + { + "id": "2607.05174", + "title": "AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments", + "url": "https://arxiv.org/abs/2607.05174", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Zhiheng Xi", + "Dingwen Yang", + "Jiaqi Liu", + "Jixuan Huang", + "Honglin Guo", + "Baodai Huang", + "Tinggang Chen", + "Qi Zhang", + "Zhonghang Lu", + "Chenyu Liu", + "Jiajun Sun", + "Jiazheng Zhang", + "Dingwei Zhu", + "Xin Guo", + "Junzhe Wang", + "Zhihao Zhang", + "Yuming Yang", + "Junjie Ye", + "Minghe Gao", + "Dongrui Liu", + "Jiaming Ji", + "Guohao Li", + "Tao Gui", + "Qi Zhang", + "Xuanjing Huang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent", + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.05174", + "source": "arxiv", + "source_id": "arxiv:2607.05174", + "pdf_url": "https://arxiv.org/pdf/2607.05174", + "primary_query": "language-agent" + }, + { + "id": "2607.05029", + "title": "Your Agent's Memories Are Not Its Own: Forged Reasoning Attacks on LLM Agent Memory and Defenses", + "url": "https://arxiv.org/abs/2607.05029", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Neeraj Karamchandani", + "Piyush Nagasubramaniam", + "Sencun Zhu", + "Dinghao Wu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "llm-agent" + ], + "arxiv_id": "2607.05029", + "source": "arxiv", + "source_id": "arxiv:2607.05029", + "pdf_url": "https://arxiv.org/pdf/2607.05029", + "primary_query": "agent-memory" + }, + { + "id": "2607.05120", + "title": "Agent Data Injection Attacks are Realistic Threats to AI Agents", + "url": "https://arxiv.org/abs/2607.05120", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Woohyuk Choi", + "Juhee Kim", + "Taehyun Kang", + "Jihyeon Jeong", + "Luyi Xing", + "Byoungyoung Lee" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "ai-agent", + "coding-agent", + "web-gui-agent" + ], + "arxiv_id": "2607.05120", + "source": "arxiv", + "source_id": "arxiv:2607.05120", + "pdf_url": "https://arxiv.org/pdf/2607.05120", + "primary_query": "agent-safety" + }, + { + "id": "2607.05202", + "title": "EvoAgentBench: Benchmarking Agent Self-Evolution via Ability Transfer", + "url": "https://arxiv.org/abs/2607.05202", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Xingze Gao", + "Chuanrui Hu", + "Hongda Chen", + "Pengfei Yao", + "Zhao Wang", + "Yi Bai", + "Zhengwei Wu", + "Yunyun Han", + "Xiaofeng Cong", + "Jie Gui", + "Yafeng Deng", + "Teng Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "planning", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2607.05202", + "source": "arxiv", + "source_id": "arxiv:2607.05202", + "pdf_url": "https://arxiv.org/pdf/2607.05202", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.04395", + "title": "NKI-Agent: Domain-Specific Fine-Tuning and Agentic Tool Use for Neuron Kernel Generation", + "url": "https://arxiv.org/abs/2607.04395", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Junjie Tang", + "Jun Huan", + "Hao Zhou", + "Yuhao Zhang", + "Lin Wang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.04395", + "source": "arxiv", + "source_id": "arxiv:2607.04395", + "pdf_url": "https://arxiv.org/pdf/2607.04395", + "primary_query": "tool-use" + }, + { + "id": "2607.03510", + "title": "CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI", + "url": "https://arxiv.org/abs/2607.03510", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Roopam W. Sure" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.03510", + "source": "arxiv", + "source_id": "arxiv:2607.03510", + "pdf_url": "https://arxiv.org/pdf/2607.03510", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02684", + "title": "Can Coding Agents Implement Missed Compiler Optimizations? Evaluating LLM Agents on LLVM Peephole Optimizations", + "url": "https://arxiv.org/abs/2607.02684", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Hongxu Xu", + "Chunhao Liao", + "Xintong Zhou", + "Chengnian Sun" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.02684", + "source": "arxiv", + "source_id": "arxiv:2607.02684", + "pdf_url": "https://arxiv.org/pdf/2607.02684", + "primary_query": "coding-agent" + }, + { + "id": "2607.02507", + "title": "What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates", + "url": "https://arxiv.org/abs/2607.02507", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Arman Ghaffarizadeh", + "Danyal Mohaddes", + "Aliakbar Izadkhah", + "Shahriar Noroozizadeh" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.02507", + "source": "arxiv", + "source_id": "arxiv:2607.02507", + "pdf_url": "https://arxiv.org/pdf/2607.02507", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.30986", + "title": "The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows", + "url": "https://arxiv.org/abs/2606.30986", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Canhui Liu" + ], + "categories": [ + "cs.CY", + "cs.HC", + "cs.MA", + "econ.GN" + ], + "topics": [ + "memory", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent" + ], + "arxiv_id": "2606.30986", + "source": "arxiv", + "source_id": "arxiv:2606.30986", + "pdf_url": "https://arxiv.org/pdf/2606.30986", + "primary_query": "agentic-ai" + }, + { + "id": "2606.30639", + "title": "Self-Evolving World Models for LLM Agent Planning", + "url": "https://arxiv.org/abs/2606.30639", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Xuan Zhang", + "Wenxuan Zhang", + "See-Kiong Ng", + "Yang Deng" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.30639", + "source": "arxiv", + "source_id": "arxiv:2606.30639", + "pdf_url": "https://arxiv.org/pdf/2606.30639", + "primary_query": "llm-agent" + }, + { + "id": "2606.29771", + "title": "CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents", + "url": "https://arxiv.org/abs/2606.29771", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Bo Qu", + "Mingguang Chen" + ], + "categories": [ + "cs.AI", + "cs.LG", + "q-fin.CP", + "q-fin.PM" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29771", + "source": "arxiv", + "source_id": "arxiv:2606.29771", + "pdf_url": "https://arxiv.org/pdf/2606.29771", + "primary_query": "llm-agent" + }, + { + "id": "2607.00041", + "title": "ATM: CID-Brokered Pre-Write Admission for Multi-Agent Code Co-Synthesis", + "url": "https://arxiv.org/abs/2607.00041", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Eagl Huang" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "planning", + "rag" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "multi-agent-llm" + ], + "arxiv_id": "2607.00041", + "source": "arxiv", + "source_id": "arxiv:2607.00041", + "pdf_url": "https://arxiv.org/pdf/2607.00041", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.29774", + "title": "Analytic Concept-Centric Memory for Agentic Embodied Manipulation", + "url": "https://arxiv.org/abs/2606.29774", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Mingyang Sun", + "Xiujian Liang", + "Jiude Wei", + "Qichen He", + "Donglin Wang", + "Cewu Lu", + "Jianhua Sun" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.29774", + "source": "arxiv", + "source_id": "arxiv:2606.29774", + "pdf_url": "https://arxiv.org/pdf/2606.29774", + "primary_query": "agent-memory" + }, + { + "id": "2606.29193", + "title": "A Multi-Dataset Benchmark for Evaluating LLM Agents in Microservice Failure Diagnosis", + "url": "https://arxiv.org/abs/2606.29193", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Yuanhong Cai", + "Xiaohui Nie", + "Kanglin Yin", + "Changhua Pei", + "Yongqian Sun", + "Shenglin Zhang", + "Haibin Liu", + "Guiyang Liu", + "Xidao Wen", + "Fang Situ", + "Dan Pei" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29193", + "source": "arxiv", + "source_id": "arxiv:2606.29193", + "pdf_url": "https://arxiv.org/pdf/2606.29193", + "primary_query": "llm-agent" + }, + { + "id": "2606.29030", + "title": "Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering", + "url": "https://arxiv.org/abs/2606.29030", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Shahnewaz Karim Sakib", + "Anindya Bijoy Das" + ], + "categories": [ + "cs.AI", + "cs.ET" + ], + "topics": [ + "memory", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.29030", + "source": "arxiv", + "source_id": "arxiv:2606.29030", + "pdf_url": "https://arxiv.org/pdf/2606.29030", + "primary_query": "ai-agent" + }, + { + "id": "2606.28692", + "title": "An AI agent for treatment reasoning over a biomedical tool universe", + "url": "https://arxiv.org/abs/2606.28692", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Shanghua Gao", + "Ayush Noori", + "Richard Zhu", + "Curtis Ginder", + "Zhenglun Kong", + "Xiaorui Su", + "Justin Kauffman", + "Benjamin S. Glicksberg", + "Joshua Lampert", + "Ankit Sakhuja", + "Ashwin Sawant", + "ATHENA-R1 Evaluation Consortium", + "David A. Clifton", + "Noa Dagan", + "Ran Balicer", + "Marinka Zitnik" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "tool-use" + ], + "arxiv_id": "2606.28692", + "source": "arxiv", + "source_id": "arxiv:2606.28692", + "pdf_url": "https://arxiv.org/pdf/2606.28692", + "primary_query": "ai-agent" + }, + { + "id": "2606.27806", + "title": "Agent vs. Parametric World Models: Hybrid Planning for Reliable Language Agents", + "url": "https://arxiv.org/abs/2606.27806", + "published": "2026-06-26", + "updated": "2026-07-05", + "authors": [ + "Xinyuan Song", + "Zekun Cai" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.27806", + "source": "arxiv", + "source_id": "arxiv:2606.27806", + "pdf_url": "https://arxiv.org/pdf/2606.27806", + "primary_query": "language-agent" + }, + { + "id": "2606.28467", + "title": "An Agentic AI Pipeline for Appliance-Level Energy Anomaly Detection and LLM-Driven Recommendations", + "url": "https://arxiv.org/abs/2606.28467", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Dihia Falouz", + "Aida Douaibia", + "Amine Bechar", + "Youssef Elmir", + "Abbes Amira", + "Adel Oulefki" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "rag-agent" + ], + "arxiv_id": "2606.28467", + "source": "arxiv", + "source_id": "arxiv:2606.28467", + "pdf_url": "https://arxiv.org/pdf/2606.28467", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26960", + "title": "Toward Agentic SysAdmin: Rethinking System Administration with AI Agents", + "url": "https://arxiv.org/abs/2606.26960", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Gianmaria Frigo", + "Davide Saladino", + "Alberto Castagnaro", + "Francesco Marchiori", + "Denis Donadel", + "Luca Pajola", + "Mauro Conti" + ], + "categories": [ + "cs.NI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.26960", + "source": "arxiv", + "source_id": "arxiv:2606.26960", + "pdf_url": "https://arxiv.org/pdf/2606.26960", + "primary_query": "ai-agent" + }, + { + "id": "2606.26346", + "title": "How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?", + "url": "https://arxiv.org/abs/2606.26346", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "David Akinpelu", + "Akintonde Abbas", + "Rereloluwa Alimi", + "Ayodeji Lana" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.26346", + "source": "arxiv", + "source_id": "arxiv:2606.26346", + "pdf_url": "https://arxiv.org/pdf/2606.26346", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.26403", + "title": "ProfileFoundry: A Synthetic Person-Object Substrate for Privacy, Memory, and Tool-Use Evaluation in LLM Agent", + "url": "https://arxiv.org/abs/2606.26403", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Sriram Selvam", + "Anneswa Ghosh" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.26403", + "source": "arxiv", + "source_id": "arxiv:2606.26403", + "pdf_url": "https://arxiv.org/pdf/2606.26403", + "primary_query": "tool-use" + }, + { + "id": "2606.25161", + "title": "TRUSTMEM: Learning Trustworthy Memory Consolidation for LLM Agents with Long-Term Memory", + "url": "https://arxiv.org/abs/2606.25161", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Tianyu Yang", + "Sudipta Paul", + "Vijay Srinivasan", + "Vivek Kulkarni", + "Srinivas Chappidi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.25161", + "source": "arxiv", + "source_id": "arxiv:2606.25161", + "pdf_url": "https://arxiv.org/pdf/2606.25161", + "primary_query": "agent-memory" + }, + { + "id": "2606.24626", + "title": "SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation", + "url": "https://arxiv.org/abs/2606.24626", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Chenyang Zhu", + "Jiayu Yao", + "Kushal Chawla", + "Youbing Yin", + "Nathan Wolfe", + "Pengshan Cai", + "Jingyu Wu", + "Spencer Hong", + "Sangwoo Cho", + "Shi-Xiong Zhang", + "Daben Liu", + "Sambit Sahu", + "Erin Babinsky" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "multi-agent-llm" + ], + "arxiv_id": "2606.24626", + "source": "arxiv", + "source_id": "arxiv:2606.24626", + "pdf_url": "https://arxiv.org/pdf/2606.24626", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.23991", + "title": "Critique of Agent Model", + "url": "https://arxiv.org/abs/2606.23991", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Eric Xing", + "Mingkai Deng", + "Jinyu Hou" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.MA", + "cs.RO" + ], + "topics": [ + "agent-safety", + "coding-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2606.23991", + "source": "arxiv", + "source_id": "arxiv:2606.23991", + "pdf_url": "https://arxiv.org/pdf/2606.23991", + "primary_query": "agentic-ai" + }, + { + "id": "2606.22844", + "title": "RaMem: Contextual Reinstatement for Long-term Agentic Memory", + "url": "https://arxiv.org/abs/2606.22844", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Wei Yang", + "Bryce Kan", + "Shixuan Li", + "Li Li", + "Yuehan Qin", + "Jiate Li", + "Paul Bogdan", + "Jesse Thomason" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.22844", + "source": "arxiv", + "source_id": "arxiv:2606.22844", + "pdf_url": "https://arxiv.org/pdf/2606.22844", + "primary_query": "agent-memory" + }, + { + "id": "2606.23565", + "title": "HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory", + "url": "https://arxiv.org/abs/2606.23565", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Xiaolin Zhou", + "Liu Liu", + "Tingyang Xiao", + "Wei Feng", + "Fa Fu", + "Xinrui Meng", + "Xinjie Wang", + "Jialiang Han", + "Boyang Yu", + "Yun Du", + "Wei Sui", + "Zhizhong Su" + ], + "categories": [ + "cs.RO", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.23565", + "source": "arxiv", + "source_id": "arxiv:2606.23565", + "pdf_url": "https://arxiv.org/pdf/2606.23565", + "primary_query": "planning-agent" + }, + { + "id": "2606.22678", + "title": "RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents", + "url": "https://arxiv.org/abs/2606.22678", + "published": "2026-06-21", + "updated": "2026-06-29", + "authors": [ + "Meher Bhaskar Madiraju", + "Meher Sai Preetam Madiraju" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.22678", + "source": "arxiv", + "source_id": "arxiv:2606.22678", + "pdf_url": "https://arxiv.org/pdf/2606.22678", + "primary_query": "coding-agent" + }, + { + "id": "2606.22417", + "title": "Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent", + "url": "https://arxiv.org/abs/2606.22417", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Ishaan Bhola", + "Adithyan Krishnan", + "Sravanth Kurmala", + "Mukunda NS" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.22417", + "source": "arxiv", + "source_id": "arxiv:2606.22417", + "pdf_url": "https://arxiv.org/pdf/2606.22417", + "primary_query": "coding-agent" + }, + { + "id": "2606.21129", + "title": "AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents", + "url": "https://arxiv.org/abs/2606.21129", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Zhen Zhao", + "Yu Zhang", + "Yanpeng Zhu", + "Jia Wang", + "Songqiao Tao", + "Xin Cheng", + "Jiexin Gao" + ], + "categories": [ + "cs.CR", + "cs.OS" + ], + "topics": [ + "agent-safety", + "planning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "autonomous-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.21129", + "source": "arxiv", + "source_id": "arxiv:2606.21129", + "pdf_url": "https://arxiv.org/pdf/2606.21129", + "primary_query": "ai-agent" + }, + { + "id": "2606.21649", + "title": "EvoEmbedding: Evolvable Representations for Long-Context Retrieval and Agentic Memory", + "url": "https://arxiv.org/abs/2606.21649", + "published": "2026-06-19", + "updated": "2026-06-25", + "authors": [ + "Chang Nie", + "Chaoyou Fu", + "Junlan Feng", + "Caifeng Shan" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "agentic-ai", + "rag-agent" + ], + "arxiv_id": "2606.21649", + "source": "arxiv", + "source_id": "arxiv:2606.21649", + "pdf_url": "https://arxiv.org/pdf/2606.21649", + "primary_query": "agent-memory" + }, + { + "id": "2606.20950", + "title": "Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering", + "url": "https://arxiv.org/abs/2606.20950", + "published": "2026-06-18", + "updated": "2026-07-02", + "authors": [ + "Sergei Trashchenkov" + ], + "categories": [ + "cs.AI", + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "ai-agent", + "tool-use" + ], + "arxiv_id": "2606.20950", + "source": "arxiv", + "source_id": "arxiv:2606.20950", + "pdf_url": "https://arxiv.org/pdf/2606.20950", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.19704", + "title": "Beyond Static Leaderboards: Predictive Validity for the Evaluation of LLM Agents", + "url": "https://arxiv.org/abs/2606.19704", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Dhaval C. Patel", + "Kaoutar El Maghraoui", + "Shuxin Lin", + "Yusheng Li", + "Tianjun Feng", + "Chun-Yi Tsai", + "Yihan Sun", + "Wei Alexander Xin", + "Akshat Bhandari", + "Tanisha Rathod", + "Aaron Fan", + "Sanskruti Vijay Shejwal", + "Tomas Pasiecznik", + "Sagar Chethan Kumar", + "Tanmay Agarwal", + "Rohith Kanathur", + "Sam Colman", + "Amaan Sheikh", + "Dev Bahl", + "Ann Li", + "Krish Veera", + "Alimurtaza Mustafa Merchant", + "Shambhawi Baswaraj Bhure", + "Sajal Kumar Goyla", + "Chengrui Li", + "Kirthana Natarajan", + "Rui Li", + "Thomas Ajai", + "Rujing Li", + "Vivek G. Iyer", + "Sanjaii Vijayakumar", + "Yitong Bai", + "Ayal Yakobe", + "Darief Maes", + "Yassine Jebbouri", + "Tianyang Xu", + "Thai Quoc On", + "Vera Mazeeva", + "Winston Li", + "Yuval Shemla", + "Yeshitha Bhuvanesh", + "Rushin Bhatt", + "Siddharth Chethan Gowda", + "Alisha Vinod", + "Caroline Cahill", + "Shriya Aishani Rachakonda", + "Yunfeng Chen", + "Aryaman Agrawal", + "Aman Upganlawar", + "Mao Le Jonathan Ang", + "Yubin Sally Go", + "Madhav Rajkondawar", + "Yang-Jung Chen", + "Trisha Maturi", + "Ananya Kapoor", + "Andrew Li", + "Shrey Arora", + "Mana Abbaszadeh", + "Shen Li", + "Charles Xu", + "Byeolah Kwon" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.19704", + "source": "arxiv", + "source_id": "arxiv:2606.19704", + "pdf_url": "https://arxiv.org/pdf/2606.19704", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.20512", + "title": "Probe-and-Refine Tuning of Repository Guidance for Coding Agents", + "url": "https://arxiv.org/abs/2606.20512", + "published": "2026-06-18", + "updated": "2026-06-19", + "authors": [ + "Asa Shepard", + "Jeannie Albrecht" + ], + "categories": [ + "cs.SE", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "tool-use" + ], + "arxiv_id": "2606.20512", + "source": "arxiv", + "source_id": "arxiv:2606.20512", + "pdf_url": "https://arxiv.org/pdf/2606.20512", + "primary_query": "coding-agent" + }, + { + "id": "2606.18829", + "title": "GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents", + "url": "https://arxiv.org/abs/2606.18829", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Zhe Ren", + "Yibo Yang", + "Yimeng Chen", + "Zijun Zhao", + "Benshuo Fu", + "Zhihao Shu", + "Bingjie Zhang", + "Yangyang Xu", + "Dandan Guo", + "Shuicheng Yan" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.18829", + "source": "arxiv", + "source_id": "arxiv:2606.18829", + "pdf_url": "https://arxiv.org/pdf/2606.18829", + "primary_query": "agent-memory" + }, + { + "id": "2606.18356", + "title": "SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.18356", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Yuchuan Tian", + "Mengyu Zheng", + "Haocheng Mei", + "Ye Yuan", + "Chao Xu", + "Xinghao Chen", + "Hanting Chen", + "Yu Wang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "tool-use" + ], + "arxiv_id": "2606.18356", + "source": "arxiv", + "source_id": "arxiv:2606.18356", + "pdf_url": "https://arxiv.org/pdf/2606.18356", + "primary_query": "agent-safety" + }, + { + "id": "2606.16774", + "title": "OpenClaw-Skill: Collective Skill Tree Search for Agentic Large Language Models", + "url": "https://arxiv.org/abs/2606.16774", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Tianyi Lin", + "Chuanyu Sun", + "Jingyi Zhang", + "Changxu Wei", + "Huanjin Yao", + "Shunyu Liu", + "Xikun Zhang", + "Liu Liu", + "Jiaxing Huang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent", + "tool-use" + ], + "arxiv_id": "2606.16774", + "source": "arxiv", + "source_id": "arxiv:2606.16774", + "pdf_url": "https://arxiv.org/pdf/2606.16774", + "primary_query": "planning-agent" + }, + { + "id": "2606.15862", + "title": "RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments", + "url": "https://arxiv.org/abs/2606.15862", + "published": "2026-06-14", + "updated": "2026-06-19", + "authors": [ + "Linghua Zhang", + "Jun Wang", + "Jingtong Wu", + "Zhisong Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.15862", + "source": "arxiv", + "source_id": "arxiv:2606.15862", + "pdf_url": "https://arxiv.org/pdf/2606.15862", + "primary_query": "tool-use" + }, + { + "id": "2606.12586", + "title": "Beyond Attack Success Rate: Examining Trigger Leakage in Vision-Language Agentic Systems", + "url": "https://arxiv.org/abs/2606.12586", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Jiamin Chang", + "Salil Kanhere", + "Piotr Koniusz", + "Jason", + "Xue", + "Hammond Pearce" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent", + "tool-use" + ], + "arxiv_id": "2606.12586", + "source": "arxiv", + "source_id": "arxiv:2606.12586", + "pdf_url": "https://arxiv.org/pdf/2606.12586", + "primary_query": "language-agent" + }, + { + "id": "2606.10507", + "title": "HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning", + "url": "https://arxiv.org/abs/2606.10507", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Juncheng Diao", + "Zhicong Lu", + "Peiguang Li", + "Yongwei Zhou", + "Changyuan Tian", + "Qingbin Li", + "Rongxiang Weng", + "Jingang Wang", + "Xunliang Cai" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2606.10507", + "source": "arxiv", + "source_id": "arxiv:2606.10507", + "pdf_url": "https://arxiv.org/pdf/2606.10507", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.11042", + "title": "Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields", + "url": "https://arxiv.org/abs/2606.11042", + "published": "2026-06-09", + "updated": "2026-06-11", + "authors": [ + "Liya Zhu", + "Jingzhe Ding", + "Jian Zhang", + "Jianbo Xue", + "Shihao Liang", + "Ge Zhang", + "Yi Zhu", + "Duju Zeng", + "Xiang Gao", + "Qingshui Gu", + "Mailun Gao", + "Huimin Che", + "Yan Zhao", + "Peiheng Zhou", + "Haojun Wang", + "Chaobo Xian", + "Lili Le", + "Chi Wu", + "Yiwei Liu", + "Shengda Long", + "Jiale Yang", + "Fangzhi Xu", + "Sijin Wu", + "Haodong Duan", + "Chao He", + "Zhaojian Li", + "Minchao Wang", + "Huan Zhou", + "Jiani Hou", + "Chuqian Yu", + "Weiran Shi", + "Hongwan Gao", + "Jiamin Chen", + "Guanhong Chen", + "Tingqin Luo", + "Kaiyuan Zhang", + "Zhixin Yao", + "Qing Hua", + "Yuhao Jiang", + "Jin Chen", + "Pu Chen", + "Zhenyu Hu", + "Xingyu Li", + "Zhengxuan Jiang", + "Meng Cao", + "Tianfeng Long", + "Haozhe Wang", + "Mingzhang Wang", + "Yichen Zhang", + "Yiming Dai", + "Chenchen Zhang", + "Jiaying Wang", + "Xinying Liu", + "Xingzu Liu", + "Lingling Zhang", + "Xinjie Chen", + "Yujia Qin", + "Wangchunshu Zhou", + "Zhiyong Wu", + "Yang Liu", + "Jiaheng Liu", + "Lei Zhang", + "Shen Yan", + "Wenhao Huang", + "Zaiyuan Wang", + "Xiaolong Chang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.11042", + "source": "arxiv", + "source_id": "arxiv:2606.11042", + "pdf_url": "https://arxiv.org/pdf/2606.11042", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09483", + "title": "Memory Beyond Recall: A Dual-Process Cognitive Memory System for Self-Evolving LLM Agents", + "url": "https://arxiv.org/abs/2606.09483", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Tianxiang Fei", + "Mingyang Song", + "Mao Zheng", + "Xiang Yu" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.09483", + "source": "arxiv", + "source_id": "arxiv:2606.09483", + "pdf_url": "https://arxiv.org/pdf/2606.09483", + "primary_query": "agent-memory" + }, + { + "id": "2606.07314", + "title": "QBugLM: An Agentic Benchmarking Framework for LLM-based Quantum Software Debugging", + "url": "https://arxiv.org/abs/2606.07314", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "An B. B. Pham", + "Hoa T. Nguyen", + "Muhammad Usman" + ], + "categories": [ + "cs.SE", + "cs.ET", + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "reasoning", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.07314", + "source": "arxiv", + "source_id": "arxiv:2606.07314", + "pdf_url": "https://arxiv.org/pdf/2606.07314", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.05684", + "title": "AdaMEM: Test-Time Adaptive Memory for Language Agents", + "url": "https://arxiv.org/abs/2606.05684", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Yunxiang Zhang", + "Yiheng Li", + "Ali Payani", + "Lu Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "language-agent" + ], + "arxiv_id": "2606.05684", + "source": "arxiv", + "source_id": "arxiv:2606.05684", + "pdf_url": "https://arxiv.org/pdf/2606.05684", + "primary_query": "agent-memory" + }, + { + "id": "2606.06448", + "title": "Agent Memory: Characterization and System Implications of Stateful Long-Horizon Workloads", + "url": "https://arxiv.org/abs/2606.06448", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Yasmine Omri", + "Ziyu Gan", + "Zachary Broveak", + "Robin Geens", + "Zexue He", + "Alex Pentland", + "Marian Verhelst", + "Tsachy Weissman", + "Thierry Tambe" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.06448", + "source": "arxiv", + "source_id": "arxiv:2606.06448", + "pdf_url": "https://arxiv.org/pdf/2606.06448", + "primary_query": "agent-memory" + }, + { + "id": "2606.04874", + "title": "Agent Planning Benchmark: A Diagnostic Framework for Planning Capabilities in LLM Agents", + "url": "https://arxiv.org/abs/2606.04874", + "published": "2026-06-03", + "updated": "2026-06-05", + "authors": [ + "Haoyu Sun", + "Wenxuan Wang", + "Mingyang Song", + "Jujie He", + "Weinan Zhang", + "Yang Liu", + "Yang Yang", + "Yu Cheng" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "planning-agent" + ], + "arxiv_id": "2606.04874", + "source": "arxiv", + "source_id": "arxiv:2606.04874", + "pdf_url": "https://arxiv.org/pdf/2606.04874", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.28349", + "title": "HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning", + "url": "https://arxiv.org/abs/2606.28349", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Zeju Li", + "Ziyang Zheng", + "Yizhou Zhou", + "Qiang Xu" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.28349", + "source": "arxiv", + "source_id": "arxiv:2606.28349", + "pdf_url": "https://arxiv.org/pdf/2606.28349", + "primary_query": "agent-memory" + }, + { + "id": "2606.04315", + "title": "Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline", + "url": "https://arxiv.org/abs/2606.04315", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Zhikai Chen", + "Jialiang Gu", + "Junyu Yin", + "Xianxuan Long", + "Shenglai Zeng", + "Xiaoze Liu", + "Kai Guo", + "Keren Zhou", + "Jiliang Tang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04315", + "source": "arxiv", + "source_id": "arxiv:2606.04315", + "pdf_url": "https://arxiv.org/pdf/2606.04315", + "primary_query": "agent-memory" + }, + { + "id": "2606.03374", + "title": "eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents", + "url": "https://arxiv.org/abs/2606.03374", + "published": "2026-06-02", + "updated": "2026-06-22", + "authors": [ + "A. Haroon Rasheed", + "Maria Kabtoul" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2606.03374", + "source": "arxiv", + "source_id": "arxiv:2606.03374", + "pdf_url": "https://arxiv.org/pdf/2606.03374", + "primary_query": "agent-memory" + }, + { + "id": "2606.02372", + "title": "COMAP: Co-Evolving World Models and Agent Policies for LLM Agents", + "url": "https://arxiv.org/abs/2606.02372", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Youwei Liu", + "Jian Wang", + "Hanlin Wang", + "Wenjie Li" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent", + "planning-agent" + ], + "arxiv_id": "2606.02372", + "source": "arxiv", + "source_id": "arxiv:2606.02372", + "pdf_url": "https://arxiv.org/pdf/2606.02372", + "primary_query": "language-agent" + }, + { + "id": "2606.01613", + "title": "TechRAG: Evidence-Gated Multimodal Agentic RAG for Technical Literature Reasoning", + "url": "https://arxiv.org/abs/2606.01613", + "published": "2026-06-01", + "updated": "2026-06-13", + "authors": [ + "Kanwar Bharat Singh" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.01613", + "source": "arxiv", + "source_id": "arxiv:2606.01613", + "pdf_url": "https://arxiv.org/pdf/2606.01613", + "primary_query": "rag-agent" + }, + { + "id": "2606.00939", + "title": "FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation", + "url": "https://arxiv.org/abs/2606.00939", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Chao Peter Yang", + "Zixiao Tan", + "Kaisen Yao", + "Ziyu Zhou", + "Eleanor Jiang", + "Michael Wu" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.00939", + "source": "arxiv", + "source_id": "arxiv:2606.00939", + "pdf_url": "https://arxiv.org/pdf/2606.00939", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.30690", + "title": "ElasticMem: Latent Memory as a Learnable Resource for LLM Agents", + "url": "https://arxiv.org/abs/2605.30690", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Tao Feng", + "Chongrui Ye", + "Tianyang Luo", + "Jingjun Xu", + "Xueqiang Xu", + "Haozhen Zhang", + "Ge Liu", + "Jiaxuan You" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.30690", + "source": "arxiv", + "source_id": "arxiv:2605.30690", + "pdf_url": "https://arxiv.org/pdf/2605.30690", + "primary_query": "planning-agent" + }, + { + "id": "2606.20629", + "title": "Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams", + "url": "https://arxiv.org/abs/2606.20629", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Yinsicheng Jiang", + "Liang Cheng", + "Yeqi Huang", + "Yufan Zhao", + "Zhan Lu", + "Li Dong", + "Wenda Li", + "Edoardo Ponti", + "Luo Mai" + ], + "categories": [ + "cs.MA", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.20629", + "source": "arxiv", + "source_id": "arxiv:2606.20629", + "pdf_url": "https://arxiv.org/pdf/2606.20629", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.30604", + "title": "An Organization-Scoped LLM Agent Runtime Architecture for Regulated Cybersecurity Operations", + "url": "https://arxiv.org/abs/2605.30604", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "George Fatouros", + "Georgios Makridis", + "George Kousiouris", + "John Soldatos", + "Dimosthenis Kyriazis" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.30604", + "source": "arxiv", + "source_id": "arxiv:2605.30604", + "pdf_url": "https://arxiv.org/pdf/2605.30604", + "primary_query": "planning-agent" + }, + { + "id": "2605.27240", + "title": "ENPMR-Bench: Benchmarking Proactive Memory Retrieval for Emotional Support Agents", + "url": "https://arxiv.org/abs/2605.27240", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Xing Fu", + "Yulin Hu", + "Mengtong Ji", + "Haozhen Li", + "Yixin Sun", + "Weixiang Zhao", + "Yanyan Zhao", + "Bing Qin" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.27240", + "source": "arxiv", + "source_id": "arxiv:2605.27240", + "pdf_url": "https://arxiv.org/pdf/2605.27240", + "primary_query": "language-agent" + }, + { + "id": "2605.25200", + "title": "GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning", + "url": "https://arxiv.org/abs/2605.25200", + "published": "2026-05-24", + "updated": "2026-06-03", + "authors": [ + "Xiang Cheng", + "Yulan Hu", + "Lulu Zheng", + "Zheng Pan", + "Xin Li", + "Yong Liu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.25200", + "source": "arxiv", + "source_id": "arxiv:2605.25200", + "pdf_url": "https://arxiv.org/pdf/2605.25200", + "primary_query": "planning-agent" + }, + { + "id": "2605.24220", + "title": "Polar: Agentic RL on Any Harness at Scale", + "url": "https://arxiv.org/abs/2605.24220", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Binfeng Xu", + "Hao Zhang", + "Shaokun Zhang", + "Songyang Han", + "Mingjie Liu", + "Jian Hu", + "Shizhe Diao", + "Zhenghui Jin", + "Yunheng Zou", + "Michael Demoret", + "Jan Kautz", + "Yi Dong" + ], + "categories": [ + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.24220", + "source": "arxiv", + "source_id": "arxiv:2605.24220", + "pdf_url": "https://arxiv.org/pdf/2605.24220", + "primary_query": "language-agent" + }, + { + "id": "2605.23574", + "title": "Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2605.23574", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Yuandao Cai", + "Yuzhang Zhu", + "Liyou Gao", + "Wensheng Tang", + "Shengchao Qin" + ], + "categories": [ + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.23574", + "source": "arxiv", + "source_id": "arxiv:2605.23574", + "pdf_url": "https://arxiv.org/pdf/2605.23574", + "primary_query": "language-agent" + }, + { + "id": "2605.24069", + "title": "When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents", + "url": "https://arxiv.org/abs/2605.24069", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Shi Liu", + "Xuehai Tang", + "Xikang Yang", + "Liang Lin", + "Biyu Zhou", + "Wenjie Xiao", + "Wantao Liu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.24069", + "source": "arxiv", + "source_id": "arxiv:2605.24069", + "pdf_url": "https://arxiv.org/pdf/2605.24069", + "primary_query": "planning-agent" + }, + { + "id": "2605.24216", + "title": "Agent-ToM: Learning to Monitor Autonomous LLM Agents via Theory-of-Mind Reasoning", + "url": "https://arxiv.org/abs/2605.24216", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Nesreen K. Ahmed", + "Nima Nafisi" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.24216", + "source": "arxiv", + "source_id": "arxiv:2605.24216", + "pdf_url": "https://arxiv.org/pdf/2605.24216", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.18421", + "title": "EvoMemBench: Benchmarking Agent Memory from a Self-Evolving Perspective", + "url": "https://arxiv.org/abs/2605.18421", + "published": "2026-05-18", + "updated": "2026-06-15", + "authors": [ + "Yuyao Wang", + "Zhongjian Zhang", + "Mo Chi", + "Kaichi Yu", + "Yuhan Li", + "Miao Peng", + "Bing Tong", + "Chen Zhang", + "Yan Zhou", + "Jia Li" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "planning-agent" + ], + "arxiv_id": "2605.18421", + "source": "arxiv", + "source_id": "arxiv:2605.18421", + "pdf_url": "https://arxiv.org/pdf/2605.18421", + "primary_query": "agent-memory" + }, + { + "id": "2605.10779", + "title": "LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments", + "url": "https://arxiv.org/abs/2605.10779", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Chiyu Zhang", + "Huiqin Yang", + "Bendong Jiang", + "Xiaolei Zhang", + "Yiran Zhao", + "Ruyi Chen", + "Lu Zhou", + "Xiaogang Xu", + "Jiafei Wu", + "Liming Fang", + "Zhe Liu" + ], + "categories": [ + "cs.CR", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.10779", + "source": "arxiv", + "source_id": "arxiv:2605.10779", + "pdf_url": "https://arxiv.org/pdf/2605.10779", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.03312", + "title": "MemFlow: Intent-Driven Memory Orchestration for Small Language Model Agents", + "url": "https://arxiv.org/abs/2605.03312", + "published": "2026-05-05", + "updated": "2026-05-05", + "authors": [ + "Jiayi Chen", + "Yingcong Li", + "Guiling Wang" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.03312", + "source": "arxiv", + "source_id": "arxiv:2605.03312", + "pdf_url": "https://arxiv.org/pdf/2605.03312", + "primary_query": "language-agent" + }, + { + "id": "2605.01101", + "title": "Virtual Speech Therapist: A Clinician-in-the-Loop AI Speech Therapy Agent for Personalized and Supervised Therapy", + "url": "https://arxiv.org/abs/2605.01101", + "published": "2026-05-01", + "updated": "2026-06-15", + "authors": [ + "Shakeel Sheikh", + "Patrick Marmaroli", + "MD Sahidullah", + "Slim Ouni", + "Fabrice Hirsch", + "Goncalo Leal", + "Bjorn W Schuller" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.SD", + "eess.AS" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.01101", + "source": "arxiv", + "source_id": "arxiv:2605.01101", + "pdf_url": "https://arxiv.org/pdf/2605.01101", + "primary_query": "planning-agent" + }, + { + "id": "2604.19844", + "title": "If you're waiting for a sign... that might not be it! Mitigating Trust Boundary Confusion from Visual Injections on Vision-Language Agentic Systems", + "url": "https://arxiv.org/abs/2604.19844", + "published": "2026-04-21", + "updated": "2026-04-21", + "authors": [ + "Jiamin Chang", + "Minhui Xue", + "Ruoxi Sun", + "Shuchao Pang", + "Salil S. Kanhere", + "Hammond Pearce" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "multi-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.19844", + "source": "arxiv", + "source_id": "arxiv:2604.19844", + "pdf_url": "https://arxiv.org/pdf/2604.19844", + "primary_query": "language-agent" + }, + { + "id": "2604.18658", + "title": "Owner-Harm: A Missing Threat Model for AI Agent Safety", + "url": "https://arxiv.org/abs/2604.18658", + "published": "2026-04-20", + "updated": "2026-04-20", + "authors": [ + "Dongcheng Zhang", + "Yiqing Jiang" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.18658", + "source": "arxiv", + "source_id": "arxiv:2604.18658", + "pdf_url": "https://arxiv.org/pdf/2604.18658", + "primary_query": "agent-safety" + }, + { + "id": "2604.17562", + "title": "SafeAgent: A Runtime Protection Architecture for Agentic Systems", + "url": "https://arxiv.org/abs/2604.17562", + "published": "2026-04-19", + "updated": "2026-04-19", + "authors": [ + "Hailin Liu", + "Eugene Ilyushin", + "Jie Ni", + "Min Zhu" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.17562", + "source": "arxiv", + "source_id": "arxiv:2604.17562", + "pdf_url": "https://arxiv.org/pdf/2604.17562", + "primary_query": "agent-safety" + }, + { + "id": "2603.00623", + "title": "TraceSIR: A Multi-Agent Framework for Structured Analysis and Reporting of Agentic Execution Traces", + "url": "https://arxiv.org/abs/2603.00623", + "published": "2026-02-28", + "updated": "2026-02-28", + "authors": [ + "Shu-Xun Yang", + "Cunxiang Wang", + "Haoke Zhang", + "Wenbo Yu", + "Lindong Wu", + "Jiayi Gui", + "Dayong Yang", + "Yukuo Cen", + "Zhuoer Feng", + "Bosi Wen", + "Yidong Wang", + "Lucen Zhong", + "Jiamin Ren", + "Linfeng Zhang", + "Jie Tang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.00623", + "source": "arxiv", + "source_id": "arxiv:2603.00623", + "pdf_url": "https://arxiv.org/pdf/2603.00623", + "primary_query": "function-calling" + }, + { + "id": "2602.13530", + "title": "REMem: Reasoning with Episodic Memory in Language Agent", + "url": "https://arxiv.org/abs/2602.13530", + "published": "2026-02-13", + "updated": "2026-02-28", + "authors": [ + "Yiheng Shu", + "Saisri Padmaja Jonnalagedda", + "Xiang Gao", + "Bernal Jiménez Gutiérrez", + "Weijian Qi", + "Kamalika Das", + "Huan Sun", + "Yu Su" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.13530", + "source": "arxiv", + "source_id": "arxiv:2602.13530", + "pdf_url": "https://arxiv.org/pdf/2602.13530", + "primary_query": "language-agent" + }, + { + "id": "2510.03847", + "title": "Small Language Models for Agentic Systems: A Survey of Architectures, Capabilities, and Deployment Trade offs", + "url": "https://arxiv.org/abs/2510.03847", + "published": "2025-10-04", + "updated": "2025-10-04", + "authors": [ + "Raghav Sharma", + "Manan Mehta" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 19, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.03847", + "source": "arxiv", + "source_id": "arxiv:2510.03847", + "pdf_url": "https://arxiv.org/pdf/2510.03847", + "primary_query": "function-calling" + }, + { + "id": "2607.06157", + "title": "LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability", + "url": "https://arxiv.org/abs/2607.06157", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Chenxu Wang", + "Yongkun Yang", + "Boyuan Du", + "Shiwei Lin", + "Huaping Liu" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.06157", + "source": "arxiv", + "source_id": "arxiv:2607.06157", + "pdf_url": "https://arxiv.org/pdf/2607.06157", + "primary_query": "llm-agent" + }, + { + "id": "2607.04686", + "title": "ToolFailBench: Diagnosing Tool-Use Failures in LLM Agents", + "url": "https://arxiv.org/abs/2607.04686", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Harsh Soni" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.04686", + "source": "arxiv", + "source_id": "arxiv:2607.04686", + "pdf_url": "https://arxiv.org/pdf/2607.04686", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04240", + "title": "Biological Motifs for Agentic Control", + "url": "https://arxiv.org/abs/2607.04240", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Bogdan Banu" + ], + "categories": [ + "cs.AI", + "q-bio.CB" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm", + "multi-agent-llm" + ], + "arxiv_id": "2607.04240", + "source": "arxiv", + "source_id": "arxiv:2607.04240", + "pdf_url": "https://arxiv.org/pdf/2607.04240", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.03441", + "title": "No Time Like the Present: Agentic Test-Time Training for LLM Agents", + "url": "https://arxiv.org/abs/2607.03441", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Yanbo Wang", + "Jinhua Hao", + "Yuze Shi", + "Kun Yuan", + "Ming Sun" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.03441", + "source": "arxiv", + "source_id": "arxiv:2607.03441", + "pdf_url": "https://arxiv.org/pdf/2607.03441", + "primary_query": "coding-agent" + }, + { + "id": "2607.01874", + "title": "SkillCoach: Self-Evolving Rubrics for Evaluating and Enhancing Agentic Skill-Use", + "url": "https://arxiv.org/abs/2607.01874", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Jiayin Zhu", + "Kelong Mao", + "Yudong Guo", + "Dengbo He", + "Sulong Xu", + "Simiu Gu", + "Yutao Yue" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01874", + "source": "arxiv", + "source_id": "arxiv:2607.01874", + "pdf_url": "https://arxiv.org/pdf/2607.01874", + "primary_query": "llm-agent" + }, + { + "id": "2607.01793", + "title": "Safety Testing LLM Agents at Scale: From Risk Discovery to Evidence-Grounded Verification", + "url": "https://arxiv.org/abs/2607.01793", + "published": "2026-07-02", + "updated": "2026-07-04", + "authors": [ + "Yunhao Feng", + "Ruixiao Lin", + "Ming Wen", + "Qinqin He", + "Yanming Guo", + "Yifan Ding", + "Yutao Wu", + "Jialuo Chen", + "Zhuoer Xu", + "Xiaohu Du", + "Jianan Ma", + "Zixing Chen", + "Xingjun Ma", + "Yunhao Chen", + "Xinhao Deng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01793", + "source": "arxiv", + "source_id": "arxiv:2607.01793", + "pdf_url": "https://arxiv.org/pdf/2607.01793", + "primary_query": "llm-agent" + }, + { + "id": "2607.02703", + "title": "LLMoxie: Exploring Agentic AI for Scientific Software Development", + "url": "https://arxiv.org/abs/2607.02703", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Landung Setiawan", + "Anant Mittal", + "Cordero Core", + "Anshul Tambay", + "Carlos Garcia Jurado Suarez", + "David A. C. Beck", + "Andrew J. Connolly", + "Vani Mandava" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.DC", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2607.02703", + "source": "arxiv", + "source_id": "arxiv:2607.02703", + "pdf_url": "https://arxiv.org/pdf/2607.02703", + "primary_query": "agentic-ai" + }, + { + "id": "2607.00627", + "title": "AGI Maze as a Benchmark Framework for World-Modeling Agents", + "url": "https://arxiv.org/abs/2607.00627", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Alexey Potapov" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00627", + "source": "arxiv", + "source_id": "arxiv:2607.00627", + "pdf_url": "https://arxiv.org/pdf/2607.00627", + "primary_query": "llm-agent" + }, + { + "id": "2607.02606", + "title": "ChainSWE: Benchmarking Coding Agents on Multi-Bug Software Maintenance", + "url": "https://arxiv.org/abs/2607.02606", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Qirui Jin", + "Lingching Tung", + "Kenan Li", + "Qiyang Shi", + "Yushi She", + "Huanzhong Jia", + "Harrison Zhao", + "Kejing Xia", + "Zhenbang Du", + "Yikai Zhang", + "Jiaxin Pei", + "Zhenyu Zhang", + "Zhen Qi", + "Yuyan Duan", + "Wenke Lee", + "Zijian Jin" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02606", + "source": "arxiv", + "source_id": "arxiv:2607.02606", + "pdf_url": "https://arxiv.org/pdf/2607.02606", + "primary_query": "coding-agent" + }, + { + "id": "2606.32025", + "title": "Generative Skill Composition for LLM Agents", + "url": "https://arxiv.org/abs/2606.32025", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Xinyu Zhao", + "Zhen Tan", + "Vaishnav Tadiparthi", + "Nakul Agarwal", + "Kwonjoon Lee", + "Ehsan Moradi Pari", + "Hossein Nourkhiz Mahjoub", + "Tianlong Chen" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.32025", + "source": "arxiv", + "source_id": "arxiv:2606.32025", + "pdf_url": "https://arxiv.org/pdf/2606.32025", + "primary_query": "coding-agent" + }, + { + "id": "2606.31229", + "title": "Agentic-Ideation: Sample Efficient Agentic Trajectories Synthesis for Scientific Ideation Agents", + "url": "https://arxiv.org/abs/2606.31229", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Keyu Zhao", + "Lingyan Kong", + "Fengli Xu", + "Yong Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.31229", + "source": "arxiv", + "source_id": "arxiv:2606.31229", + "pdf_url": "https://arxiv.org/pdf/2606.31229", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31410", + "title": "Xiaomi-GUI-0 Technical Report", + "url": "https://arxiv.org/abs/2606.31410", + "published": "2026-06-30", + "updated": "2026-07-01", + "authors": [ + "Wanxia Cao", + "Chengzhen Duan", + "Pei Fu", + "Pengzhi Gao", + "Niu Lian", + "Fazhan Liu", + "Hui Liu", + "Heng Qu", + "Qinzhuo Wu", + "Zhehao Yu", + "Tongbo Chen", + "Shiqi Cui", + "Anan Du", + "Shukai Jia", + "Yuanfa Li", + "Wei Liu", + "Yike Liu", + "Wenchao Lu", + "Zhenbo Luo", + "Haoyuan Sun", + "Jiatong Sun", + "Cheng Tan", + "Yajie Wang", + "Changqiao Wu", + "Tao Xiong", + "Jiahui Yang", + "Yuxuan Yuan", + "Ruoceng Zhang", + "Shaojie Zhang", + "Jian Zhu", + "Jian Luan", + "Cong Zou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.31410", + "source": "arxiv", + "source_id": "arxiv:2606.31410", + "pdf_url": "https://arxiv.org/pdf/2606.31410", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.29178", + "title": "Selective Memory Retention for Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2606.29178", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Pranath Reddy" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29178", + "source": "arxiv", + "source_id": "arxiv:2606.29178", + "pdf_url": "https://arxiv.org/pdf/2606.29178", + "primary_query": "llm-agent" + }, + { + "id": "2606.28456", + "title": "Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game", + "url": "https://arxiv.org/abs/2606.28456", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Subhendu Bhandary", + "Federico Carucci", + "Christos Charalambous", + "Francesca Dilisante", + "Ksenia Dvorkina", + "Anna Garbo", + "Jiaqi Liang", + "Riccardo Vasellini", + "Francesco Bertolotti" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-safety", + "memory", + "multi-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.28456", + "source": "arxiv", + "source_id": "arxiv:2606.28456", + "pdf_url": "https://arxiv.org/pdf/2606.28456", + "primary_query": "ai-agent" + }, + { + "id": "2606.27406", + "title": "Towards Evaluation of Implicit Software World Models in Coding LLMs", + "url": "https://arxiv.org/abs/2606.27406", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Egor Bogomolov", + "Yaroslav Zharov" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "reasoning", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2606.27406", + "source": "arxiv", + "source_id": "arxiv:2606.27406", + "pdf_url": "https://arxiv.org/pdf/2606.27406", + "primary_query": "ai-agent" + }, + { + "id": "2606.26627", + "title": "Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents", + "url": "https://arxiv.org/abs/2606.26627", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Nada Lahjouji", + "Ashwin Gerard Colaco" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.26627", + "source": "arxiv", + "source_id": "arxiv:2606.26627", + "pdf_url": "https://arxiv.org/pdf/2606.26627", + "primary_query": "agent-memory" + }, + { + "id": "2606.26479", + "title": "Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents", + "url": "https://arxiv.org/abs/2606.26479", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Praneeth Narisetty", + "Shiva Nagendra Babu Kore", + "Uday Kumar Reddy Kattamanchi", + "Jayaram Kumarapu" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.26479", + "source": "arxiv", + "source_id": "arxiv:2606.26479", + "pdf_url": "https://arxiv.org/pdf/2606.26479", + "primary_query": "tool-use" + }, + { + "id": "2606.26793", + "title": "MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG", + "url": "https://arxiv.org/abs/2606.26793", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Inderjeet Singh", + "Andrés Murillo", + "Motoyoshi Sekiya", + "Yuki Unno", + "Junichi Suga" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.26793", + "source": "arxiv", + "source_id": "arxiv:2606.26793", + "pdf_url": "https://arxiv.org/pdf/2606.26793", + "primary_query": "rag-agent" + }, + { + "id": "2606.27472", + "title": "Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents", + "url": "https://arxiv.org/abs/2606.27472", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Vedant Patel" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.27472", + "source": "arxiv", + "source_id": "arxiv:2606.27472", + "pdf_url": "https://arxiv.org/pdf/2606.27472", + "primary_query": "planning-agent" + }, + { + "id": "2606.25622", + "title": "Probabilistic Agents in Deterministic Audits: Evaluating Multi-Agent Systems for Automated Audits Based on the German IT-Grundschutz", + "url": "https://arxiv.org/abs/2606.25622", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Lea Roxanne Muth", + "Marian Margraf" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25622", + "source": "arxiv", + "source_id": "arxiv:2606.25622", + "pdf_url": "https://arxiv.org/pdf/2606.25622", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.25358", + "title": "Agentic Knowledge Tracing: A Multi-Agent LLM Architecture for Stealth Assessment of Financial Literacy in Serious Games", + "url": "https://arxiv.org/abs/2606.25358", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Gabriel Santos", + "Rita Julia", + "Marcelo Nascimento" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25358", + "source": "arxiv", + "source_id": "arxiv:2606.25358", + "pdf_url": "https://arxiv.org/pdf/2606.25358", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.25334", + "title": "Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care", + "url": "https://arxiv.org/abs/2606.25334", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Runwei Guan", + "Yi Zhou", + "Heyi Lin", + "Jinjing Zhu", + "Mingyuan Hou", + "Yang Yang", + "Fang Yuan", + "Xiaohong Lin", + "Shaofeng Liang", + "Xuming Hu", + "Tao Li", + "Tianbin Zhao", + "Yutao Yue", + "Zhiyuan Wang", + "Hui Xiong" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "rag" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25334", + "source": "arxiv", + "source_id": "arxiv:2606.25334", + "pdf_url": "https://arxiv.org/pdf/2606.25334", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22673", + "title": "AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent", + "url": "https://arxiv.org/abs/2606.22673", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Weidi Luo", + "Qiming Zhang", + "Yihao Quan", + "Mingyu Jin", + "Jie Cai", + "Chaowei Xiao", + "Jingcheng Niu", + "Zhen Xiang" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "coding-agent" + ], + "arxiv_id": "2606.22673", + "source": "arxiv", + "source_id": "arxiv:2606.22673", + "pdf_url": "https://arxiv.org/pdf/2606.22673", + "primary_query": "agent-safety" + }, + { + "id": "2606.21710", + "title": "PrivacyAlign: Contextual Privacy Alignment for LLM Agents", + "url": "https://arxiv.org/abs/2606.21710", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Manveer Singh Tamber", + "Abhay Puri", + "Marc-Etienne Brunet", + "Perouz Taslakian", + "Jimmy Lin", + "Spandana Gella" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.21710", + "source": "arxiv", + "source_id": "arxiv:2606.21710", + "pdf_url": "https://arxiv.org/pdf/2606.21710", + "primary_query": "ai-agent" + }, + { + "id": "2606.21013", + "title": "Agentic Time Machine as an Infrastructure for Future-Event Forecasting", + "url": "https://arxiv.org/abs/2606.21013", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Jingyi Chai", + "Bingyang Zheng", + "Xiangrui Liu", + "Hao Lu", + "Zihang Zhou", + "Tianchen Wang", + "Kemeng Zhang", + "Siheng Chen" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.21013", + "source": "arxiv", + "source_id": "arxiv:2606.21013", + "pdf_url": "https://arxiv.org/pdf/2606.21013", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.21123", + "title": "A Multi-Agent Audit Framework for High-Stakes Reasoning: Evaluation and Interpretability in Clinical Mental Health Screening", + "url": "https://arxiv.org/abs/2606.21123", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Jingchen Ye", + "Yanpei Yu", + "Luyao Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.21123", + "source": "arxiv", + "source_id": "arxiv:2606.21123", + "pdf_url": "https://arxiv.org/pdf/2606.21123", + "primary_query": "rag-agent" + }, + { + "id": "2606.19899", + "title": "Measuring Biological Capabilities and Risks of AI Agents", + "url": "https://arxiv.org/abs/2606.19899", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Patricia Paskov", + "Jeffrey Lee", + "Kyle Brady", + "Alyssa Worland" + ], + "categories": [ + "cs.CY", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "agentic-ai", + "ai-agent" + ], + "arxiv_id": "2606.19899", + "source": "arxiv", + "source_id": "arxiv:2606.19899", + "pdf_url": "https://arxiv.org/pdf/2606.19899", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.20243", + "title": "Phoenix: Safe GitHub Issue Resolution via Multi-Agent LLMs", + "url": "https://arxiv.org/abs/2606.20243", + "published": "2026-06-18", + "updated": "2026-06-22", + "authors": [ + "Kipngeno Koech", + "Muhammad Adam", + "Baimam Boukar Jean Jacques", + "Joao Barros" + ], + "categories": [ + "cs.SE", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "multi-agent", + "planning", + "rag" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.20243", + "source": "arxiv", + "source_id": "arxiv:2606.20243", + "pdf_url": "https://arxiv.org/pdf/2606.20243", + "primary_query": "coding-agent" + }, + { + "id": "2606.18950", + "title": "RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models", + "url": "https://arxiv.org/abs/2606.18950", + "published": "2026-06-17", + "updated": "2026-06-18", + "authors": [ + "San Kim", + "Daechul Ahn", + "Reokyoung Kim", + "Hyeonbeom Choi", + "Seungyeon Jwa", + "Jonghyun Choi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.18950", + "source": "arxiv", + "source_id": "arxiv:2606.18950", + "pdf_url": "https://arxiv.org/pdf/2606.18950", + "primary_query": "agent-memory" + }, + { + "id": "2606.16659", + "title": "FraudSMSWalker: Benchmarking Agentic Large Language Models for SMS-to-Webpage Fraud Detection", + "url": "https://arxiv.org/abs/2606.16659", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Y. H. Zhou", + "Z. M. Ma", + "Y. J. Zhou", + "Y. T. Li", + "H. X. Xiang", + "Y. M. Cheng", + "T. L. Chen", + "K. J. Zhang", + "Z. H. Nan", + "J. H. Ni", + "Z. Wu", + "Q. Y. Pan", + "S. Zhang", + "S. Cheng", + "M. Y. Luo" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.16659", + "source": "arxiv", + "source_id": "arxiv:2606.16659", + "pdf_url": "https://arxiv.org/pdf/2606.16659", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.17041", + "title": "Benchmarking LLM Agents on Meta-Analysis Articles from Nature Portfolio", + "url": "https://arxiv.org/abs/2606.17041", + "published": "2026-06-15", + "updated": "2026-07-01", + "authors": [ + "Anzhe Xie", + "Weihang Su", + "Yujia Zhou", + "Yiqun Liu", + "Qingyao Ai" + ], + "categories": [ + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.17041", + "source": "arxiv", + "source_id": "arxiv:2606.17041", + "pdf_url": "https://arxiv.org/pdf/2606.17041", + "primary_query": "rag-agent" + }, + { + "id": "2606.16576", + "title": "Can LLM Agents Infer World Models? Evidence from Agentic Automata Learning", + "url": "https://arxiv.org/abs/2606.16576", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Reef Menaged", + "Gili Lior", + "Shauli Ravfogel", + "Roee Aharoni", + "Gabriel Stanovsky" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.16576", + "source": "arxiv", + "source_id": "arxiv:2606.16576", + "pdf_url": "https://arxiv.org/pdf/2606.16576", + "primary_query": "planning-agent" + }, + { + "id": "2606.12780", + "title": "ProPlay: Procedural World Models for Self-Evolving LLM Agents", + "url": "https://arxiv.org/abs/2606.12780", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Yijun Ma", + "Zehong Wang", + "Yiyang Li", + "Ziming Li", + "Xiaoguang Guo", + "Weixiang Sun", + "Chuxu Zhang", + "Yanfang Ye" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "tool-use", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.12780", + "source": "arxiv", + "source_id": "arxiv:2606.12780", + "pdf_url": "https://arxiv.org/pdf/2606.12780", + "primary_query": "planning-agent" + }, + { + "id": "2606.12320", + "title": "A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents", + "url": "https://arxiv.org/abs/2606.12320", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Krti Tallam" + ], + "categories": [ + "cs.AI", + "cs.CC", + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.12320", + "source": "arxiv", + "source_id": "arxiv:2606.12320", + "pdf_url": "https://arxiv.org/pdf/2606.12320", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.11680", + "title": "Organize then Retrieve: Hierarchical Memory Navigation for Efficient Agents", + "url": "https://arxiv.org/abs/2606.11680", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Hao-Lun Hsu", + "Nikki Lijing Kuang", + "Boyi Liu", + "Zhewei Yao", + "Yuxiong He" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.11680", + "source": "arxiv", + "source_id": "arxiv:2606.11680", + "pdf_url": "https://arxiv.org/pdf/2606.11680", + "primary_query": "agent-memory" + }, + { + "id": "2606.12657", + "title": "TrajGenAgent: A Hierarchical LLM Agent for Human Mobility Trajectory Generation", + "url": "https://arxiv.org/abs/2606.12657", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Siyu Li", + "Toan Tran", + "Lingyi Zhao", + "Khurram Shafique", + "Li Xiong" + ], + "categories": [ + "cs.AI", + "cs.DB", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "workflow-agent", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.12657", + "source": "arxiv", + "source_id": "arxiv:2606.12657", + "pdf_url": "https://arxiv.org/pdf/2606.12657", + "primary_query": "planning-agent" + }, + { + "id": "2606.10677", + "title": "Infini Memory: Maintainable Topic Documents for Long-Term LLM Agent Memory", + "url": "https://arxiv.org/abs/2606.10677", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Suozhao Ji", + "Baodong Wu", + "Zehao Wang", + "Lei Xia", + "Qingping Li", + "Ruisong Wang", + "Wenbo Ding", + "Zhenhua Zhu", + "Boxun Li", + "Guohao Dai", + "Yu Wang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.10677", + "source": "arxiv", + "source_id": "arxiv:2606.10677", + "pdf_url": "https://arxiv.org/pdf/2606.10677", + "primary_query": "agent-memory" + }, + { + "id": "2606.11354", + "title": "A Zero-Shot Multi-Agent Framework for Human-Building Interaction via Programmatic Reasoning", + "url": "https://arxiv.org/abs/2606.11354", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yuqi Wang", + "Gulai Shen", + "Ali Mehmani" + ], + "categories": [ + "cs.ET" + ], + "topics": [ + "agent-safety", + "coding-agent", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.11354", + "source": "arxiv", + "source_id": "arxiv:2606.11354", + "pdf_url": "https://arxiv.org/pdf/2606.11354", + "primary_query": "rag-agent" + }, + { + "id": "2606.05558", + "title": "Autoregressive Diffusion World Models for Off-Policy Evaluation of LLM Agents", + "url": "https://arxiv.org/abs/2606.05558", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Kaixuan Liu", + "Guojun Xiong", + "Weinan Zhang", + "Shengpu Tang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.05558", + "source": "arxiv", + "source_id": "arxiv:2606.05558", + "pdf_url": "https://arxiv.org/pdf/2606.05558", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.04555", + "title": "Temporal Order Matters for Agentic Memory: Segment Trees for Long-Horizon Agents", + "url": "https://arxiv.org/abs/2606.04555", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Yifan Simon Liu", + "Liam Gallagher", + "Faeze Moradi Kalarde", + "Jiazhou Liang", + "Armin Toroghi", + "Scott Sanner" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04555", + "source": "arxiv", + "source_id": "arxiv:2606.04555", + "pdf_url": "https://arxiv.org/pdf/2606.04555", + "primary_query": "agent-memory" + }, + { + "id": "2606.03895", + "title": "Agent libOS: A Runtime Substrate for Capability-Controlled Self-Evolving LLM Agents", + "url": "https://arxiv.org/abs/2606.03895", + "published": "2026-06-02", + "updated": "2026-06-29", + "authors": [ + "Yingqi Zhang" + ], + "categories": [ + "cs.OS", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.03895", + "source": "arxiv", + "source_id": "arxiv:2606.03895", + "pdf_url": "https://arxiv.org/pdf/2606.03895", + "primary_query": "planning-agent" + }, + { + "id": "2606.02302", + "title": "SeClaw: Spec-Driven Security Task Synthesis for Evaluating Autonomous Agents", + "url": "https://arxiv.org/abs/2606.02302", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Hao Cheng", + "Changtao Miao", + "Tianle Song", + "Yin Wu", + "He Liu", + "Erjia Xiao", + "Junchi Chen", + "Xiaoyu Shi", + "Yichi Wang", + "Jing Yang", + "Taowen Wang", + "Jinhao Duan", + "Mengshu Sun", + "Peiyan Dong", + "Xuan Shen", + "Yang Cao", + "Renjing Xu", + "Kaidi Xu", + "Jindong Gu", + "Bo Zhang", + "Jize Zhang", + "Chenhao Lin", + "Philip Torr", + "Chao Shen" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.02302", + "source": "arxiv", + "source_id": "arxiv:2606.02302", + "pdf_url": "https://arxiv.org/pdf/2606.02302", + "primary_query": "agent-safety" + }, + { + "id": "2605.30711", + "title": "SAGE: A Novelty Gate for Efficient Memory Evolution in Agentic LLMs", + "url": "https://arxiv.org/abs/2605.30711", + "published": "2026-05-29", + "updated": "2026-06-18", + "authors": [ + "Sijia Wang", + "Dhanajit Brahma", + "Ricardo Henao" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG", + "stat.ML" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.30711", + "source": "arxiv", + "source_id": "arxiv:2605.30711", + "pdf_url": "https://arxiv.org/pdf/2605.30711", + "primary_query": "agent-memory" + }, + { + "id": "2605.29341", + "title": "WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction", + "url": "https://arxiv.org/abs/2605.29341", + "published": "2026-05-28", + "updated": "2026-06-01", + "authors": [ + "Chengzhi Liu", + "Yuzhe Yang", + "Sophia Xiao Pu", + "Yepeng Liu", + "Lin Long", + "Yichen Guo", + "Nuo Chen", + "Zhaotian Weng", + "Elena Kochkina", + "Simerjot Kaur", + "Charese Smiley", + "Xiaomo Liu", + "James Zou", + "Sheng Liu", + "Yuheng Bu", + "Songyou Peng", + "Xin Eric Wang" + ], + "categories": [ + "cs.CV", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2605.29341", + "source": "arxiv", + "source_id": "arxiv:2605.29341", + "pdf_url": "https://arxiv.org/pdf/2605.29341", + "primary_query": "agent-memory" + }, + { + "id": "2605.27690", + "title": "TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling", + "url": "https://arxiv.org/abs/2605.27690", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Jiaqian Li", + "Yanshu Li", + "Boxuan Zhang", + "Ruixiang Tang", + "Kuan-Hao Huang" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.27690", + "source": "arxiv", + "source_id": "arxiv:2605.27690", + "pdf_url": "https://arxiv.org/pdf/2605.27690", + "primary_query": "agent-safety" + }, + { + "id": "2605.25141", + "title": "LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support", + "url": "https://arxiv.org/abs/2605.25141", + "published": "2026-05-24", + "updated": "2026-05-24", + "authors": [ + "Pavan Manjunath", + "Thomas Pruefer" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.25141", + "source": "arxiv", + "source_id": "arxiv:2605.25141", + "pdf_url": "https://arxiv.org/pdf/2605.25141", + "primary_query": "planning-agent" + }, + { + "id": "2605.19952", + "title": "Rethinking How to Remember: Beyond Atomic Facts in Lifelong LLM Agent Memory", + "url": "https://arxiv.org/abs/2605.19952", + "published": "2026-05-19", + "updated": "2026-05-19", + "authors": [ + "Jingwei Sun", + "Jianing Zhu", + "Jiangchao Yao", + "Tongliang Liu", + "Bo Han" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.19952", + "source": "arxiv", + "source_id": "arxiv:2605.19952", + "pdf_url": "https://arxiv.org/pdf/2605.19952", + "primary_query": "agent-memory" + }, + { + "id": "2605.17625", + "title": "Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents", + "url": "https://arxiv.org/abs/2605.17625", + "published": "2026-05-17", + "updated": "2026-05-17", + "authors": [ + "Nikola Milosevic" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.17625", + "source": "arxiv", + "source_id": "arxiv:2605.17625", + "pdf_url": "https://arxiv.org/pdf/2605.17625", + "primary_query": "agent-memory" + }, + { + "id": "2605.16821", + "title": "Multi-Paradigm Agent Interaction in Practice:A Systematic Analysis of Generator-Evaluator, ReAct Loop,and Adversarial Evaluation in the buddyMe Framework", + "url": "https://arxiv.org/abs/2605.16821", + "published": "2026-05-16", + "updated": "2026-05-16", + "authors": [ + "Xiaohua Wang", + "Chao Han", + "Kai Yu", + "XiaoLiang Xu", + "Liang Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.16821", + "source": "arxiv", + "source_id": "arxiv:2605.16821", + "pdf_url": "https://arxiv.org/pdf/2605.16821", + "primary_query": "planning-agent" + }, + { + "id": "2605.16481", + "title": "Visual Agentic Memory: Enabling Online Long Video Understanding via Online Indexing, Hierarchical Memory, and Agentic Retrieval", + "url": "https://arxiv.org/abs/2605.16481", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Aiden Yiliu Li", + "Nels Numan", + "Anthony Steed" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.16481", + "source": "arxiv", + "source_id": "arxiv:2605.16481", + "pdf_url": "https://arxiv.org/pdf/2605.16481", + "primary_query": "agent-memory" + }, + { + "id": "2605.12061", + "title": "SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory", + "url": "https://arxiv.org/abs/2605.12061", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Juntong Wang", + "Haoyue Zhao", + "guanghui Pan", + "Xiyuan Wang", + "Yanbo Wang", + "Qiyan Deng", + "Muhan Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.12061", + "source": "arxiv", + "source_id": "arxiv:2605.12061", + "pdf_url": "https://arxiv.org/pdf/2605.12061", + "primary_query": "language-agent" + }, + { + "id": "2605.06890", + "title": "Beyond the Black Box: Interpretability of Agentic AI Tool Use", + "url": "https://arxiv.org/abs/2605.06890", + "published": "2026-05-07", + "updated": "2026-07-05", + "authors": [ + "Hariom Tatsat", + "Ariye Shater" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.06890", + "source": "arxiv", + "source_id": "arxiv:2605.06890", + "pdf_url": "https://arxiv.org/pdf/2605.06890", + "primary_query": "function-calling" + }, + { + "id": "2605.05716", + "title": "More Is Not Always Better: Cross-Component Interference in LLM Agent Scaffolding", + "url": "https://arxiv.org/abs/2605.05716", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Ming Liu" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.05716", + "source": "arxiv", + "source_id": "arxiv:2605.05716", + "pdf_url": "https://arxiv.org/pdf/2605.05716", + "primary_query": "planning-agent" + }, + { + "id": "2605.06716", + "title": "From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms", + "url": "https://arxiv.org/abs/2605.06716", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Jinghao Luo", + "Yuchen Tian", + "Chuxue Cao", + "Ziyang Luo", + "Hongzhan Lin", + "Kaixin Li", + "Chuyi Kong", + "Ruichao Yang", + "Jing Ma" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.06716", + "source": "arxiv", + "source_id": "arxiv:2605.06716", + "pdf_url": "https://arxiv.org/pdf/2605.06716", + "primary_query": "planning-agent" + }, + { + "id": "2604.25318", + "title": "Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation", + "url": "https://arxiv.org/abs/2604.25318", + "published": "2026-04-28", + "updated": "2026-04-28", + "authors": [ + "Lanshan He", + "Haozhou Pang", + "Qi Gan", + "Xin Shen", + "Ziwei Zhang", + "Yibo Liu", + "Gang Fang", + "Bo Liu", + "Kai Sheng", + "Shengfeng Zeng", + "Chaofan Li", + "Zhen Hui", + "Keer Zhou", + "Lan Zhou", + "Shujun Dai" + ], + "categories": [ + "cs.GR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2604.25318", + "source": "arxiv", + "source_id": "arxiv:2604.25318", + "pdf_url": "https://arxiv.org/pdf/2604.25318", + "primary_query": "function-calling" + }, + { + "id": "2604.25135", + "title": "FAMA: Failure-Aware Meta-Agentic Framework for Open-Source LLMs in Interactive Tool Use Environments", + "url": "https://arxiv.org/abs/2604.25135", + "published": "2026-04-28", + "updated": "2026-04-28", + "authors": [ + "Amir Saeidi", + "Venkatesh Mishra", + "Souradeep Mukhopadhyay", + "Gaowen Liu", + "Ali Payani", + "Jayanth Srinivasa", + "Chitta Baral" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.25135", + "source": "arxiv", + "source_id": "arxiv:2604.25135", + "pdf_url": "https://arxiv.org/pdf/2604.25135", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.23459", + "title": "Architecture Matters for Multi-Agent Security", + "url": "https://arxiv.org/abs/2604.23459", + "published": "2026-04-25", + "updated": "2026-04-25", + "authors": [ + "Ben Hagag", + "William L. Anderson", + "Christian Schroeder de Witt", + "Sarah Scheffler" + ], + "categories": [ + "cs.MA", + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "planning" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.23459", + "source": "arxiv", + "source_id": "arxiv:2604.23459", + "pdf_url": "https://arxiv.org/pdf/2604.23459", + "primary_query": "agent-safety" + }, + { + "id": "2604.23374", + "title": "Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents", + "url": "https://arxiv.org/abs/2604.23374", + "published": "2026-04-25", + "updated": "2026-04-25", + "authors": [ + "Yuandao Cai", + "Wensheng Tang", + "Cheng Wen", + "Shengchao Qin" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.23374", + "source": "arxiv", + "source_id": "arxiv:2604.23374", + "pdf_url": "https://arxiv.org/pdf/2604.23374", + "primary_query": "agent-safety" + }, + { + "id": "2604.22879", + "title": "Beyond Single-Agent Alignment: Preventing Context-Fragmented Violations in Multi-Agent Systems", + "url": "https://arxiv.org/abs/2604.22879", + "published": "2026-04-24", + "updated": "2026-04-24", + "authors": [ + "Jie Wu", + "Ming Gong" + ], + "categories": [ + "cs.MA", + "cs.AI", + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.22879", + "source": "arxiv", + "source_id": "arxiv:2604.22879", + "pdf_url": "https://arxiv.org/pdf/2604.22879", + "primary_query": "agent-safety" + }, + { + "id": "2604.18847", + "title": "Human-Guided Harm Recovery for Computer Use Agents", + "url": "https://arxiv.org/abs/2604.18847", + "published": "2026-04-20", + "updated": "2026-05-28", + "authors": [ + "Christy Li", + "Sky CH-Wang", + "Andi Peng", + "Andreea Bobu" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.18847", + "source": "arxiv", + "source_id": "arxiv:2604.18847", + "pdf_url": "https://arxiv.org/pdf/2604.18847", + "primary_query": "agent-safety" + }, + { + "id": "2603.18245", + "title": "Who Tests the Testers? Systematic Enumeration and Coverage Audit of LLM Agent Tool Call Safety", + "url": "https://arxiv.org/abs/2603.18245", + "published": "2026-03-18", + "updated": "2026-03-18", + "authors": [ + "Xuan Chen", + "Lu Yan", + "Ruqi Zhang", + "Xiangyu Zhang" + ], + "categories": [ + "cs.SE", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.18245", + "source": "arxiv", + "source_id": "arxiv:2603.18245", + "pdf_url": "https://arxiv.org/pdf/2603.18245", + "primary_query": "agent-safety" + }, + { + "id": "2603.07980", + "title": "\\$OneMillion-Bench: How Far are Language Agents from Human Experts?", + "url": "https://arxiv.org/abs/2603.07980", + "published": "2026-03-09", + "updated": "2026-03-09", + "authors": [ + "Qianyu Yang", + "Yang Liu", + "Jiaqi Li", + "Jun Bai", + "Hao Chen", + "Kaiyuan Chen", + "Tiliang Duan", + "Jiayun Dong", + "Xiaobo Hu", + "Zixia Jia", + "Yang Liu", + "Tao Peng", + "Yixin Ren", + "Ran Tian", + "Zaiyuan Wang", + "Yanglihong Xiao", + "Gang Yao", + "Lingyue Yin", + "Ge Zhang", + "Chun Zhang", + "Jianpeng Jiao", + "Zilong Zheng", + "Yuan Gong" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.07980", + "source": "arxiv", + "source_id": "arxiv:2603.07980", + "pdf_url": "https://arxiv.org/pdf/2603.07980", + "primary_query": "language-agent" + }, + { + "id": "2603.09002", + "title": "Security Considerations for Multi-agent Systems", + "url": "https://arxiv.org/abs/2603.09002", + "published": "2026-03-09", + "updated": "2026-04-26", + "authors": [ + "Tam Nguyen", + "Moses Ndebugre", + "Dheeraj Arremsetty" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.09002", + "source": "arxiv", + "source_id": "arxiv:2603.09002", + "pdf_url": "https://arxiv.org/pdf/2603.09002", + "primary_query": "agent-safety" + }, + { + "id": "2602.07962", + "title": "LOCA-bench: Benchmarking Language Agents Under Controllable and Extreme Context Growth", + "url": "https://arxiv.org/abs/2602.07962", + "published": "2026-02-08", + "updated": "2026-02-08", + "authors": [ + "Weihao Zeng", + "Yuzhen Huang", + "Junxian He" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.07962", + "source": "arxiv", + "source_id": "arxiv:2602.07962", + "pdf_url": "https://arxiv.org/pdf/2602.07962", + "primary_query": "language-agent" + }, + { + "id": "2602.05302", + "title": "PieArena: Ranking and Profiling Language Agents in Realistic Negotiation Scenarios", + "url": "https://arxiv.org/abs/2602.05302", + "published": "2026-02-05", + "updated": "2026-06-01", + "authors": [ + "Chris Zhu", + "Sasha Cui", + "Will Sanok Dufallo", + "Runzhi Jin", + "Zhen Xu", + "Linjun Zhang", + "Daylian Cain" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.05302", + "source": "arxiv", + "source_id": "arxiv:2602.05302", + "pdf_url": "https://arxiv.org/pdf/2602.05302", + "primary_query": "language-agent" + }, + { + "id": "2601.06007", + "title": "Don't Break the Cache: An Evaluation of Prompt Caching for Long-Horizon Agentic Tasks", + "url": "https://arxiv.org/abs/2601.06007", + "published": "2026-01-09", + "updated": "2026-01-31", + "authors": [ + "Elias Lumer", + "Faheem Nizar", + "Akshaya Jangiti", + "Kevin Frank", + "Anmol Gulati", + "Mandar Phadate", + "Vamse Kumar Subbiah" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.06007", + "source": "arxiv", + "source_id": "arxiv:2601.06007", + "pdf_url": "https://arxiv.org/pdf/2601.06007", + "primary_query": "function-calling" + }, + { + "id": "2511.04847", + "title": "Test-Time Adaptation for LLM Agents via Environment Interaction", + "url": "https://arxiv.org/abs/2511.04847", + "published": "2025-11-06", + "updated": "2026-02-22", + "authors": [ + "Arthur Chen", + "Zuxin Liu", + "Jianguo Zhang", + "Akshara Prabhakar", + "Zhiwei Liu", + "Shelby Heinecke", + "Silvio Savarese", + "Victor Zhong", + "Caiming Xiong" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2511.04847", + "source": "arxiv", + "source_id": "arxiv:2511.04847", + "pdf_url": "https://arxiv.org/pdf/2511.04847", + "primary_query": "function-calling" + }, + { + "id": "2509.10769", + "title": "AgentArch: A Comprehensive Benchmark to Evaluate Agent Architectures in Enterprise", + "url": "https://arxiv.org/abs/2509.10769", + "published": "2025-09-13", + "updated": "2026-01-06", + "authors": [ + "Tara Bogavelli", + "Roshnee Sharma", + "Hari Subramani" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 18, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.10769", + "source": "arxiv", + "source_id": "arxiv:2509.10769", + "pdf_url": "https://arxiv.org/pdf/2509.10769", + "primary_query": "function-calling" + }, + { + "id": "2607.06273", + "title": "AgentTether: Graph-Guided Diagnosis and Runtime Intervention for Reliable LLM Agent Operation", + "url": "https://arxiv.org/abs/2607.06273", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Chenyu Zhao", + "Shenglin Zhang", + "Wenwei Gu", + "Yongqian Sun", + "Dan Pei", + "Chetan Bansal", + "Saravan Rajmohan", + "Minghua Ma" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.06273", + "source": "arxiv", + "source_id": "arxiv:2607.06273", + "pdf_url": "https://arxiv.org/pdf/2607.06273", + "primary_query": "llm-agent" + }, + { + "id": "2607.06195", + "title": "LogicHunter: Testing LLM Agent Frameworks with an Agentic Oracle", + "url": "https://arxiv.org/abs/2607.06195", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Minghui Long", + "Yanjie Zhao", + "Haoyu Wang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.06195", + "source": "arxiv", + "source_id": "arxiv:2607.06195", + "pdf_url": "https://arxiv.org/pdf/2607.06195", + "primary_query": "llm-agent" + }, + { + "id": "2607.06080", + "title": "From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations", + "url": "https://arxiv.org/abs/2607.06080", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Shiyi Ling", + "Zhi Zheng", + "Hui Zheng", + "Wenjun Xue", + "Feng Ye", + "Tong Xu" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.SI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.06080", + "source": "arxiv", + "source_id": "arxiv:2607.06080", + "pdf_url": "https://arxiv.org/pdf/2607.06080", + "primary_query": "llm-agent" + }, + { + "id": "2607.05297", + "title": "MetaSkill-Evolve: Recursive Self-Improvement of LLM Agents via Two-Timescale Meta-Skill Evolution", + "url": "https://arxiv.org/abs/2607.05297", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Zefeng Wang", + "Minxi Yan", + "Jinhe Bi", + "Sikuan Yan", + "Volker Tresp", + "Yunpu Ma" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent" + ], + "arxiv_id": "2607.05297", + "source": "arxiv", + "source_id": "arxiv:2607.05297", + "pdf_url": "https://arxiv.org/pdf/2607.05297", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.04528", + "title": "Measuring Harness-Induced Belief Divergence in Multi-Step LLM Agents", + "url": "https://arxiv.org/abs/2607.04528", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Haiwen Yi", + "Xinyuan Song" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent" + ], + "arxiv_id": "2607.04528", + "source": "arxiv", + "source_id": "arxiv:2607.04528", + "pdf_url": "https://arxiv.org/pdf/2607.04528", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.04293", + "title": "CausalGame: Benchmarking Causal Thinking of LLM Agents in Games", + "url": "https://arxiv.org/abs/2607.04293", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Zhenhao Chen", + "Yongqiang Chen", + "Chenxi Liu", + "Junchi Yu", + "Xiangchen Song", + "Zijian Li", + "Jialin Li", + "Philip Torr", + "Bo Han", + "Kun Zhang" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG", + "stat.ML" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04293", + "source": "arxiv", + "source_id": "arxiv:2607.04293", + "pdf_url": "https://arxiv.org/pdf/2607.04293", + "primary_query": "llm-agent" + }, + { + "id": "2607.04426", + "title": "ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI", + "url": "https://arxiv.org/abs/2607.04426", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "ACE-Brain Team", + ":", + "Ziyang Gong", + "Haoming Gu", + "Zehang Luo", + "Tianyi Zhang", + "Tao Tao", + "Yixiao Chi", + "Zhe Liu", + "Lingsi Zhu", + "Jingyuan Liu", + "Anke Tang", + "Songze Li", + "Yilun Kong", + "Ningjing Liu", + "Tianyu Zhu", + "Yunpeng Qing", + "Shuang Luo", + "Xiang Liu", + "Shi Fu", + "Dawei Nie", + "Sixiang Liu", + "Zhexi Wen", + "Feng Pan", + "Xiaofeng Wang", + "Zhi Hou", + "Chunxiao Liu", + "Xue Yang", + "Junchi Yan", + "Hengshuang Zhao", + "Dacheng Tao", + "Xiaogang Wang" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04426", + "source": "arxiv", + "source_id": "arxiv:2607.04426", + "pdf_url": "https://arxiv.org/pdf/2607.04426", + "primary_query": "agentic-ai" + }, + { + "id": "2607.04162", + "title": "ACE: Agentic Control for Embodied Manipulation via Zero-shot Workflow Reasoning", + "url": "https://arxiv.org/abs/2607.04162", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Iok Tong Lei", + "QianZhi Li", + "Ying Jie Yap", + "Yujie Zhang", + "Rui Zhong", + "Haichao Gui", + "Xiaolong Liu", + "Zhidong Deng" + ], + "categories": [ + "cs.RO", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04162", + "source": "arxiv", + "source_id": "arxiv:2607.04162", + "pdf_url": "https://arxiv.org/pdf/2607.04162", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03333", + "title": "SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference", + "url": "https://arxiv.org/abs/2607.03333", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Huajun Bai", + "Weiwei Lv", + "Huichuan Zheng", + "Youyou Lu", + "Jiwu Shu" + ], + "categories": [ + "cs.DC", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.03333", + "source": "arxiv", + "source_id": "arxiv:2607.03333", + "pdf_url": "https://arxiv.org/pdf/2607.03333", + "primary_query": "llm-agent" + }, + { + "id": "2607.02857", + "title": "MOSAIC: Knowledge-Guided CLI Command Composition Attack in LLM Coding Agents", + "url": "https://arxiv.org/abs/2607.02857", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Jiangrong Wu", + "Huaijin Wang", + "Yihao Zhang", + "Yuhong Nan", + "Shuai Wang" + ], + "categories": [ + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2607.02857", + "source": "arxiv", + "source_id": "arxiv:2607.02857", + "pdf_url": "https://arxiv.org/pdf/2607.02857", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.01935", + "title": "A-TMA: Decoupling State-Aware Memory Failures in Long-Term Agent Memory", + "url": "https://arxiv.org/abs/2607.01935", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Zitong Shi", + "Yixuan Tang", + "Anthony Kum Hoe Tung" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "llm-agent" + ], + "arxiv_id": "2607.01935", + "source": "arxiv", + "source_id": "arxiv:2607.01935", + "pdf_url": "https://arxiv.org/pdf/2607.01935", + "primary_query": "agent-memory" + }, + { + "id": "2607.01640", + "title": "AgentFlow: Building Agent Dependency Graphs for Static Analysis of Agent Programs", + "url": "https://arxiv.org/abs/2607.01640", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Shenao Wang", + "Xinyi Hou", + "Yanjie Zhao", + "Xiao Cheng", + "Haoyu Wang" + ], + "categories": [ + "cs.SE", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.01640", + "source": "arxiv", + "source_id": "arxiv:2607.01640", + "pdf_url": "https://arxiv.org/pdf/2607.01640", + "primary_query": "llm-agent" + }, + { + "id": "2607.01668", + "title": "VeriChat: An Agentic Conversational AI Assistant for Hardware Security Verification", + "url": "https://arxiv.org/abs/2607.01668", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Dipayan Saha", + "Khan Thamid Hasan", + "Shams Tarek", + "Sujan Kumar Saha", + "Mark Tehranipoor", + "Farimah Farahmandi" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.01668", + "source": "arxiv", + "source_id": "arxiv:2607.01668", + "pdf_url": "https://arxiv.org/pdf/2607.01668", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02294", + "title": "Coding Agents Are Guessing: Measuring Action-Boundary Violations in Underspecified DevOps Instructions", + "url": "https://arxiv.org/abs/2607.02294", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Zimo Ji", + "Zekai Zhang", + "Congying Xu", + "Zongjie Li", + "Yudong Gao", + "Shuai Wang", + "Shing-Chi Cheung" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2607.02294", + "source": "arxiv", + "source_id": "arxiv:2607.02294", + "pdf_url": "https://arxiv.org/pdf/2607.02294", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.01916", + "title": "ContextSniper: AntTrail's Token-Efficient Code Memory for Repository-Level Program Repair", + "url": "https://arxiv.org/abs/2607.01916", + "published": "2026-07-02", + "updated": "2026-07-06", + "authors": [ + "Chiwang Luk", + "Matin Mohammad Najafi", + "Zhifeng Jia", + "Wei Yang", + "Xiuchang Li", + "Jinwei Zhu", + "Yang Ren", + "Lei Chen", + "Gao Cong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "coding-agent", + "rag-agent" + ], + "arxiv_id": "2607.01916", + "source": "arxiv", + "source_id": "arxiv:2607.01916", + "pdf_url": "https://arxiv.org/pdf/2607.01916", + "primary_query": "agent-memory" + }, + { + "id": "2607.01929", + "title": "Beyond Textual Repository Exploration: Dual-Modal Structural Reasoning for Agentic Issue Resolution", + "url": "https://arxiv.org/abs/2607.01929", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Jiayi Zhang", + "Kai Huang", + "Yang Liu", + "Chunyang Chen" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "function-calling" + ], + "arxiv_id": "2607.01929", + "source": "arxiv", + "source_id": "arxiv:2607.01929", + "pdf_url": "https://arxiv.org/pdf/2607.01929", + "primary_query": "coding-agent" + }, + { + "id": "2607.01071", + "title": "MemSyco-Bench: Benchmarking Sycophancy in Agent Memory", + "url": "https://arxiv.org/abs/2607.01071", + "published": "2026-07-01", + "updated": "2026-07-02", + "authors": [ + "Zhishang Xiang", + "Zerui Chen", + "Yunbo Tang", + "Zhimin Wei", + "Ruqin Ning", + "Yujie Lin", + "Qinggang Zhang", + "Jinsong Su" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2607.01071", + "source": "arxiv", + "source_id": "arxiv:2607.01071", + "pdf_url": "https://arxiv.org/pdf/2607.01071", + "primary_query": "agent-memory" + }, + { + "id": "2607.00939", + "title": "Leveraging LLM-Based Agentic Systems to Generate Quantum Applications for Test Optimization", + "url": "https://arxiv.org/abs/2607.00939", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Ming Tao", + "Yuechen Li", + "Tao Yue", + "Man Zhang", + "Aitor Arrieta Marcos" + ], + "categories": [ + "cs.SE", + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00939", + "source": "arxiv", + "source_id": "arxiv:2607.00939", + "pdf_url": "https://arxiv.org/pdf/2607.00939", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00334", + "title": "Managed Autonomy at Runtime: Gear-Based Safety and Governance for Single- and Multi-Agent Cyber-Physical Systems", + "url": "https://arxiv.org/abs/2607.00334", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Srini Ramaswamy", + "Wang Miaosheng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "multi-agent", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "multi-agent-llm" + ], + "arxiv_id": "2607.00334", + "source": "arxiv", + "source_id": "arxiv:2607.00334", + "pdf_url": "https://arxiv.org/pdf/2607.00334", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.05428", + "title": "CHARLIE: An On-Premise Multi-Agent Retrieval-Augmented Generation System for Evidential Reasoning in Forensic Science", + "url": "https://arxiv.org/abs/2607.05428", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Leandro D. Carneiro", + "Andre L. S. Meirelles", + "Juliano de A. Gomes", + "Rafael C. A. Cabral" + ], + "categories": [ + "cs.DL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.05428", + "source": "arxiv", + "source_id": "arxiv:2607.05428", + "pdf_url": "https://arxiv.org/pdf/2607.05428", + "primary_query": "rag-agent" + }, + { + "id": "2606.31693", + "title": "ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping", + "url": "https://arxiv.org/abs/2606.31693", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Jiacheng Chen", + "Tao Zhang", + "Manxi Lin", + "Dunxian Huang", + "Teng Shi", + "Honghao Fu", + "Mengyan Li", + "Xinming Zhang", + "Chenchi Zhang", + "Xuan Lu", + "Xiaoxiong Du", + "Haibin Chen", + "Shaolin Ye", + "Hao Chang", + "Xiaoqi Li", + "Shuwen Xiao", + "Yujin Yuan", + "Jingxuan Feng", + "Shaopan Xiong", + "Huimin Yi", + "Ju Huang", + "Qiu Shen", + "Ying Chen", + "Junjun Zheng", + "Xiangheng Kong", + "Yuning Jiang" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.31693", + "source": "arxiv", + "source_id": "arxiv:2606.31693", + "pdf_url": "https://arxiv.org/pdf/2606.31693", + "primary_query": "llm-agent" + }, + { + "id": "2606.31252", + "title": "Embodied CAD: Solver-Grounded LLM Agents for Parametric B-Rep Assembly Modeling", + "url": "https://arxiv.org/abs/2606.31252", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Fumin Liu", + "Haoyu Zhou", + "Fei Hao", + "Lin Yang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2606.31252", + "source": "arxiv", + "source_id": "arxiv:2606.31252", + "pdf_url": "https://arxiv.org/pdf/2606.31252", + "primary_query": "llm-agent" + }, + { + "id": "2606.31174", + "title": "ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents", + "url": "https://arxiv.org/abs/2606.31174", + "published": "2026-06-30", + "updated": "2026-07-02", + "authors": [ + "Kaiwen Xiong", + "Haonian Ji", + "Shi Qiu", + "Zeyu Zheng", + "Cihang Xie", + "Xinyu Ye", + "Huaxiu Yao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.31174", + "source": "arxiv", + "source_id": "arxiv:2606.31174", + "pdf_url": "https://arxiv.org/pdf/2606.31174", + "primary_query": "llm-agent" + }, + { + "id": "2606.31046", + "title": "OpenLife: Toward Open-World Artificial Life with Autonomous LLM Agents", + "url": "https://arxiv.org/abs/2606.31046", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Atsushi Masumori", + "Itsuki Doi", + "Norihiro Maruyama", + "Ryosuke Takata", + "Takashi Ikegami" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.31046", + "source": "arxiv", + "source_id": "arxiv:2606.31046", + "pdf_url": "https://arxiv.org/pdf/2606.31046", + "primary_query": "llm-agent" + }, + { + "id": "2606.31650", + "title": "ECHO: Prune to act, trace to learn with selective turn memory in agentic RL", + "url": "https://arxiv.org/abs/2606.31650", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Zijun Xie", + "Binbin Zheng", + "Enlei Gong", + "Jihua Liu", + "Yuyang You", + "Lingfeng Liu", + "Jiayao Tang", + "Guanqun Zhao", + "Aoqi Hu", + "Zeyu Chen" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.31650", + "source": "arxiv", + "source_id": "arxiv:2606.31650", + "pdf_url": "https://arxiv.org/pdf/2606.31650", + "primary_query": "language-agent" + }, + { + "id": "2606.31639", + "title": "A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems", + "url": "https://arxiv.org/abs/2606.31639", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Seyed Bagher Hashemi Natanzi", + "Bo Tang" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.GT", + "cs.LO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.31639", + "source": "arxiv", + "source_id": "arxiv:2606.31639", + "pdf_url": "https://arxiv.org/pdf/2606.31639", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.30566", + "title": "Forensic Trajectory Signatures for Agent Memory Poisoning Detection", + "url": "https://arxiv.org/abs/2606.30566", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Jun Wen Leong" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "llm-agent" + ], + "arxiv_id": "2606.30566", + "source": "arxiv", + "source_id": "arxiv:2606.30566", + "pdf_url": "https://arxiv.org/pdf/2606.30566", + "primary_query": "agent-memory" + }, + { + "id": "2606.29914", + "title": "MemDelta: Controlled Baselines and Hidden Confounds in Agent Memory Evaluation", + "url": "https://arxiv.org/abs/2606.29914", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Kuan Wang" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2606.29914", + "source": "arxiv", + "source_id": "arxiv:2606.29914", + "pdf_url": "https://arxiv.org/pdf/2606.29914", + "primary_query": "agent-memory" + }, + { + "id": "2606.30555", + "title": "Linguistic Firewall: Geometry as Defense in Multi-Agent Systems Routing", + "url": "https://arxiv.org/abs/2606.30555", + "published": "2026-06-29", + "updated": "2026-07-05", + "authors": [ + "Dvir Alsheich", + "Adar Peleg", + "Ben Hagag", + "Rom Himelstein", + "Amit Levi", + "Avi Mendelson" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30555", + "source": "arxiv", + "source_id": "arxiv:2606.30555", + "pdf_url": "https://arxiv.org/pdf/2606.30555", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.30546", + "title": "MAS-Lab: A Specification-Driven Validation Framework for Reliable Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.30546", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Jordan Augé", + "Giovanna Carofiglio", + "Giulio Grassi", + "Jacques Samain" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30546", + "source": "arxiv", + "source_id": "arxiv:2606.30546", + "pdf_url": "https://arxiv.org/pdf/2606.30546", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.30259", + "title": "Multi-Agentic System Leveraging Open-Source LLMs to Mitigate Disinformation Threats", + "url": "https://arxiv.org/abs/2606.30259", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Sebastian Kula", + "Martin Tamajka" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30259", + "source": "arxiv", + "source_id": "arxiv:2606.30259", + "pdf_url": "https://arxiv.org/pdf/2606.30259", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29742", + "title": "MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition", + "url": "https://arxiv.org/abs/2606.29742", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Zishan Su", + "Junjie Huang", + "Shiwen Shan", + "Xingyan Chen", + "Hui Zeng", + "Yuxin Su", + "Yanlin Wang", + "Michael R. Lyu" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.29742", + "source": "arxiv", + "source_id": "arxiv:2606.29742", + "pdf_url": "https://arxiv.org/pdf/2606.29742", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29270", + "title": "Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates", + "url": "https://arxiv.org/abs/2606.29270", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Chuan He", + "Zebin Chen", + "Zhengyi Yang", + "Shaobo Qiao", + "Mingchen Ju", + "Jiate Liu", + "Dong Wen", + "Guanfeng Liu" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.29270", + "source": "arxiv", + "source_id": "arxiv:2606.29270", + "pdf_url": "https://arxiv.org/pdf/2606.29270", + "primary_query": "llm-agent" + }, + { + "id": "2606.29654", + "title": "Budgeted Act-or-Defer Multi-Agent LLM Deliberation with Local Reliability Bounds", + "url": "https://arxiv.org/abs/2606.29654", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Mengdie Flora Wang", + "Haochen Xie", + "Guanghui Wang", + "Devin Zhang", + "Jae Oh Woo" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.29654", + "source": "arxiv", + "source_id": "arxiv:2606.29654", + "pdf_url": "https://arxiv.org/pdf/2606.29654", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28958", + "title": "When Latent Agents Lie: KV-Cache Integrity in Multi-Agent LLM Collaboration", + "url": "https://arxiv.org/abs/2606.28958", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Luís Brito", + "Carlos Baquero" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-safety", + "memory", + "multi-agent", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.28958", + "source": "arxiv", + "source_id": "arxiv:2606.28958", + "pdf_url": "https://arxiv.org/pdf/2606.28958", + "primary_query": "llm-agent" + }, + { + "id": "2606.28666", + "title": "Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare", + "url": "https://arxiv.org/abs/2606.28666", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Liam Kearns" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "rag-agent" + ], + "arxiv_id": "2606.28666", + "source": "arxiv", + "source_id": "arxiv:2606.28666", + "pdf_url": "https://arxiv.org/pdf/2606.28666", + "primary_query": "agentic-ai" + }, + { + "id": "2606.27990", + "title": "AdvancedShelLM: A Stateful Multi-Agent LLM Honeypot for SSH Deception", + "url": "https://arxiv.org/abs/2606.27990", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Muris Sladić", + "Eman Alibalić", + "Veronica Valeros", + "Carlos Catania", + "Sebastian Garcia" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.27990", + "source": "arxiv", + "source_id": "arxiv:2606.27990", + "pdf_url": "https://arxiv.org/pdf/2606.27990", + "primary_query": "llm-agent" + }, + { + "id": "2606.27929", + "title": "When Multi-Robot Systems Meet Agentic AI:Towards Embodied Collective Intelligence", + "url": "https://arxiv.org/abs/2606.27929", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Yuxuan Yan", + "Yuanyuan Jia", + "Qianqian Yang" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "multi-agent", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.27929", + "source": "arxiv", + "source_id": "arxiv:2606.27929", + "pdf_url": "https://arxiv.org/pdf/2606.27929", + "primary_query": "ai-agent" + }, + { + "id": "2606.28434", + "title": "SWE-MeM: Learning Adaptive Memory Management for Long-Horizon Coding Agents", + "url": "https://arxiv.org/abs/2606.28434", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Shuzheng Gao", + "Wenhao Zeng", + "Zhaojian Yu", + "Jianqiao Wangni", + "Chaozheng Wang", + "Kai Cai", + "Shilin He", + "Michael R. Lyu" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent", + "memory", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.28434", + "source": "arxiv", + "source_id": "arxiv:2606.28434", + "pdf_url": "https://arxiv.org/pdf/2606.28434", + "primary_query": "coding-agent" + }, + { + "id": "2606.28570", + "title": "Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG", + "url": "https://arxiv.org/abs/2606.28570", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Deep Ghosal", + "Ishani Sen", + "Wazib Ansar", + "Amlan Chakrabarti" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "rag-agent" + ], + "arxiv_id": "2606.28570", + "source": "arxiv", + "source_id": "arxiv:2606.28570", + "pdf_url": "https://arxiv.org/pdf/2606.28570", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28182", + "title": "LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior", + "url": "https://arxiv.org/abs/2606.28182", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Qinhong Zhou", + "Chuang Gan", + "Anoop Cherian" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CV", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28182", + "source": "arxiv", + "source_id": "arxiv:2606.28182", + "pdf_url": "https://arxiv.org/pdf/2606.28182", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.25514", + "title": "Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution", + "url": "https://arxiv.org/abs/2606.25514", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yang Chen", + "Aliya Ahmad", + "Yiheng Zhou", + "Reyhaneh Jabbarvand" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.25514", + "source": "arxiv", + "source_id": "arxiv:2606.25514", + "pdf_url": "https://arxiv.org/pdf/2606.25514", + "primary_query": "coding-agent" + }, + { + "id": "2606.25588", + "title": "IntentTester: Intent-Driven Multi-agent Framework for Cross-Library Test Migration", + "url": "https://arxiv.org/abs/2606.25588", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yi Gao", + "Ziyuan Zhang", + "Xing Hu", + "Xiaohu Yang", + "Xin Xia" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25588", + "source": "arxiv", + "source_id": "arxiv:2606.25588", + "pdf_url": "https://arxiv.org/pdf/2606.25588", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.25400", + "title": "BrainAgent: A Large Language Model-Driven Multi-Agent Framework for Autonomous Brain Signal Understanding", + "url": "https://arxiv.org/abs/2606.25400", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yangxuan Zhou", + "Sha Zhao", + "Jiquan Wang", + "Shijian Li", + "Gang Pan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25400", + "source": "arxiv", + "source_id": "arxiv:2606.25400", + "pdf_url": "https://arxiv.org/pdf/2606.25400", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24775", + "title": "Are We Ready For An Agent-Native Memory System?", + "url": "https://arxiv.org/abs/2606.24775", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Wei Zhou", + "Xuanhe Zhou", + "Shaokun Han", + "Hongming Xu", + "Guoliang Li", + "Zhiyu Li", + "Feiyu Xiong", + "Fan Wu" + ], + "categories": [ + "cs.CL", + "cs.DB", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.24775", + "source": "arxiv", + "source_id": "arxiv:2606.24775", + "pdf_url": "https://arxiv.org/pdf/2606.24775", + "primary_query": "agent-memory" + }, + { + "id": "2606.24649", + "title": "Agentic Collaborative Cognition for Zero-Shot 3D Understanding", + "url": "https://arxiv.org/abs/2606.24649", + "published": "2026-06-23", + "updated": "2026-06-25", + "authors": [ + "Wenxin Wang", + "Bo Zhang", + "Feng Chen", + "Zixuan Wang", + "Wen Li", + "Changsheng Li", + "Yinjie Lei" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24649", + "source": "arxiv", + "source_id": "arxiv:2606.24649", + "pdf_url": "https://arxiv.org/pdf/2606.24649", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24437", + "title": "ReM-MoA: Reasoning Memory Sustains Mixture-of-Agents Scaling", + "url": "https://arxiv.org/abs/2606.24437", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Heng Ping", + "Arijit Bhattacharjee", + "Peiyu Zhang", + "Shixuan Li", + "Wei Yang", + "Ali Jannesari", + "Nesreen Ahmed", + "Paul Bogdan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24437", + "source": "arxiv", + "source_id": "arxiv:2606.24437", + "pdf_url": "https://arxiv.org/pdf/2606.24437", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22647", + "title": "RAVEN: Agentic RAG for Automated Vulnerability Repair", + "url": "https://arxiv.org/abs/2606.22647", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Varun Gadey", + "Zijie Liu", + "Alexandra Dmitrienko" + ], + "categories": [ + "cs.CR", + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.22647", + "source": "arxiv", + "source_id": "arxiv:2606.22647", + "pdf_url": "https://arxiv.org/pdf/2606.22647", + "primary_query": "rag-agent" + }, + { + "id": "2606.22330", + "title": "Hypothesis-Driven Skill Optimization for LLM Agents", + "url": "https://arxiv.org/abs/2606.22330", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Fangxin Shang", + "Yehui Yang" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-safety", + "coding-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.22330", + "source": "arxiv", + "source_id": "arxiv:2606.22330", + "pdf_url": "https://arxiv.org/pdf/2606.22330", + "primary_query": "planning-agent" + }, + { + "id": "2606.21740", + "title": "Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents", + "url": "https://arxiv.org/abs/2606.21740", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Rajesh Mangannavar", + "Zachary Coalson", + "Pranay Dugar", + "Prasad Tadepalli" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.21740", + "source": "arxiv", + "source_id": "arxiv:2606.21740", + "pdf_url": "https://arxiv.org/pdf/2606.21740", + "primary_query": "planning-agent" + }, + { + "id": "2606.18467", + "title": "ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift", + "url": "https://arxiv.org/abs/2606.18467", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Jeffery Opoku", + "David Banahene" + ], + "categories": [ + "stat.ML", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "agentic-ai", + "ai-agent", + "rag-agent", + "tool-use" + ], + "arxiv_id": "2606.18467", + "source": "arxiv", + "source_id": "arxiv:2606.18467", + "pdf_url": "https://arxiv.org/pdf/2606.18467", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18142", + "title": "Your AI Travel Agent Would Book You a Bullfight: An Agentic Benchmark for Implicit Animal Welfare in Frontier AI Models", + "url": "https://arxiv.org/abs/2606.18142", + "published": "2026-06-16", + "updated": "2026-07-06", + "authors": [ + "Jasmine Brazilek", + "Joel Christoph", + "Maheep Chaudhary", + "Oliver Tullio", + "Carol Kline", + "Miles Tidmarsh", + "Arturs Kanepajs" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.18142", + "source": "arxiv", + "source_id": "arxiv:2606.18142", + "pdf_url": "https://arxiv.org/pdf/2606.18142", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.16591", + "title": "SING: Synthetic Intention Graph for Scalable Active Tool Discovery in LLM Agents", + "url": "https://arxiv.org/abs/2606.16591", + "published": "2026-06-15", + "updated": "2026-06-16", + "authors": [ + "Qiao Xiao", + "Haochen Shi", + "Yisen Gao", + "Wenbin Hu", + "Huihao Jing", + "Tianshi Zheng", + "Baixuan Xu", + "Ziheng Zhang", + "Weiqi Wang", + "Haoran Li", + "Jiaxin Bai", + "Yangqiu Song" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.16591", + "source": "arxiv", + "source_id": "arxiv:2606.16591", + "pdf_url": "https://arxiv.org/pdf/2606.16591", + "primary_query": "tool-use" + }, + { + "id": "2606.15684", + "title": "Multi-agent Framework for Time-Sensitive Complementary Collaboration in Minecraft", + "url": "https://arxiv.org/abs/2606.15684", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Juheon Yi", + "Jinglu Wang", + "Xiaoyi Zhang", + "Yan Lu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.15684", + "source": "arxiv", + "source_id": "arxiv:2606.15684", + "pdf_url": "https://arxiv.org/pdf/2606.15684", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.15931", + "title": "DeepRoot: A KG-Coordinated Multi-Agent System for Therapeutic Reasoning over Historical Medical Texts", + "url": "https://arxiv.org/abs/2606.15931", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Zijian Carl Ma", + "Sean J. Wang", + "Sijbren Kramer", + "Li Erran Li" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.15931", + "source": "arxiv", + "source_id": "arxiv:2606.15931", + "pdf_url": "https://arxiv.org/pdf/2606.15931", + "primary_query": "tool-use" + }, + { + "id": "2606.12703", + "title": "SMSR: Certified Defence Against Runtime Memory Poisoning in Persistent LLM Agent Systems", + "url": "https://arxiv.org/abs/2606.12703", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Tarun Sharma" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.12703", + "source": "arxiv", + "source_id": "arxiv:2606.12703", + "pdf_url": "https://arxiv.org/pdf/2606.12703", + "primary_query": "rag-agent" + }, + { + "id": "2606.10684", + "title": "Divide and Cooperate: Role-Decomposed Multi-Agent LLM Training with Cross-Agent Learning Signals", + "url": "https://arxiv.org/abs/2606.10684", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Jaewan Park", + "Solbee Cho", + "Jay-Yoon Lee" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.10684", + "source": "arxiv", + "source_id": "arxiv:2606.10684", + "pdf_url": "https://arxiv.org/pdf/2606.10684", + "primary_query": "language-agent" + }, + { + "id": "2606.10616", + "title": "Learning What to Remember: Observability-Safe Memory Retention via Constrained Optimization for Long-Horizon Language Agents", + "url": "https://arxiv.org/abs/2606.10616", + "published": "2026-06-09", + "updated": "2026-06-29", + "authors": [ + "Qingcan Kang", + "Liu Mingyang", + "Shixiong Kai", + "Kaichao Liang", + "Tao Zhong", + "Mingxuan Yuan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.10616", + "source": "arxiv", + "source_id": "arxiv:2606.10616", + "pdf_url": "https://arxiv.org/pdf/2606.10616", + "primary_query": "language-agent" + }, + { + "id": "2606.10933", + "title": "Frontier Coding Agents Use Metaprogramming to Adapt to Unfamiliar Programming Languages", + "url": "https://arxiv.org/abs/2606.10933", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Aman Sharma", + "Sushrut Thorat", + "Paras Chopra" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.10933", + "source": "arxiv", + "source_id": "arxiv:2606.10933", + "pdf_url": "https://arxiv.org/pdf/2606.10933", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.10577", + "title": "AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness", + "url": "https://arxiv.org/abs/2606.10577", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yijian Li", + "Changze Li", + "Hantian Shi", + "Jiaying Luo", + "Jiyuan Cai", + "Ming Yang", + "Tong Qin" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.10577", + "source": "arxiv", + "source_id": "arxiv:2606.10577", + "pdf_url": "https://arxiv.org/pdf/2606.10577", + "primary_query": "agent-memory" + }, + { + "id": "2606.10304", + "title": "MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents", + "url": "https://arxiv.org/abs/2606.10304", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Pratibha Revankar", + "Kargi Chauhan", + "Jihye Kim", + "Sadiba Nusrat Nur", + "Vincent Siu", + "Chenguang Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.10304", + "source": "arxiv", + "source_id": "arxiv:2606.10304", + "pdf_url": "https://arxiv.org/pdf/2606.10304", + "primary_query": "planning-agent" + }, + { + "id": "2606.07682", + "title": "SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work?", + "url": "https://arxiv.org/abs/2606.07682", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Rishi Desai", + "Jesse Hu", + "Joan Cabezas", + "Neel Harsola", + "Pratyush Shukla", + "Roey Ben Chaim", + "Adnan El Assadi", + "Omkaar Mukund Kamath", + "Fenil Faldu", + "Prannay Hebbar", + "Jiankai Sun", + "Yiyuan Li", + "Pramod Srinivasan", + "Ishan Gupta", + "Christopher Settles", + "Daniel Wang", + "Derek Chen", + "Pranav Raja", + "Albert Liu", + "Marek Šuppa", + "Nevasini Sasikumar", + "Luyang Kong", + "Erik Quintanilla", + "Xiangyi Li", + "Ivan Bercovich", + "Steven Dillmann" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.07682", + "source": "arxiv", + "source_id": "arxiv:2606.07682", + "pdf_url": "https://arxiv.org/pdf/2606.07682", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.07867", + "title": "The Cold-Start Safety Gap in LLM Agents", + "url": "https://arxiv.org/abs/2606.07867", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Chung-En Sun", + "Linbo Liu", + "Tsui-Wei Weng" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.07867", + "source": "arxiv", + "source_id": "arxiv:2606.07867", + "pdf_url": "https://arxiv.org/pdf/2606.07867", + "primary_query": "agent-safety" + }, + { + "id": "2606.07711", + "title": "Rosetta Memory: Adaptive Memory for Cross-LLM Agents", + "url": "https://arxiv.org/abs/2606.07711", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Hao Yang", + "Shiqi Shen", + "Haoxuan Li", + "Zhipeng Wang", + "Zhi Gong", + "Xu Chen" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "coding-agent", + "memory", + "planning", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.07711", + "source": "arxiv", + "source_id": "arxiv:2606.07711", + "pdf_url": "https://arxiv.org/pdf/2606.07711", + "primary_query": "planning-agent" + }, + { + "id": "2606.06054", + "title": "Beyond Similarity: Trustworthy Memory Search for Personal AI Agents", + "url": "https://arxiv.org/abs/2606.06054", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Jiawen Zhang", + "Kejia Chen", + "Jiachen Ma", + "Yangfan Hu", + "Lipeng He", + "Yechao Zhang", + "Jian Liu", + "Xiaohu Yang", + "Tianwei Zhang", + "Ruoxi Jia" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.06054", + "source": "arxiv", + "source_id": "arxiv:2606.06054", + "pdf_url": "https://arxiv.org/pdf/2606.06054", + "primary_query": "agent-memory" + }, + { + "id": "2606.05805", + "title": "From Risk Classification to Action Plan Remediation: A Guardrail Feedback Driven Framework for LLM Agents", + "url": "https://arxiv.org/abs/2606.05805", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Yuhao Sun", + "Jiacheng Zhang", + "Shaanan Cohney", + "Zhexin Zhang", + "Feng Liu", + "Xingliang Yuan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.05805", + "source": "arxiv", + "source_id": "arxiv:2606.05805", + "pdf_url": "https://arxiv.org/pdf/2606.05805", + "primary_query": "planning-agent" + }, + { + "id": "2606.04599", + "title": "Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection", + "url": "https://arxiv.org/abs/2606.04599", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Yongzi Yu", + "Ao Li", + "Le Wang", + "Ziyue Li", + "Fugee Tsung", + "Yuxuan Liang", + "Man Li" + ], + "categories": [ + "cs.AI", + "cs.CE" + ], + "topics": [ + "agent-safety", + "multi-agent", + "planning", + "rag", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.04599", + "source": "arxiv", + "source_id": "arxiv:2606.04599", + "pdf_url": "https://arxiv.org/pdf/2606.04599", + "primary_query": "planning-agent" + }, + { + "id": "2606.04051", + "title": "RUBAS: Rubric-Based Reinforcement Learning for Agent Safety", + "url": "https://arxiv.org/abs/2606.04051", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Xian Qi Loye", + "Qinglin Su", + "Zhexin Zhang", + "Shiyao Cui", + "Qi Zhu", + "Fei Mi", + "Hongning Wang", + "Minlie Huang" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.04051", + "source": "arxiv", + "source_id": "arxiv:2606.04051", + "pdf_url": "https://arxiv.org/pdf/2606.04051", + "primary_query": "agent-safety" + }, + { + "id": "2606.02812", + "title": "Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection", + "url": "https://arxiv.org/abs/2606.02812", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Sihang Zeng", + "Matthew Thompson", + "Ruth Etzioni", + "Meliha Yetisgen" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "rag", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.02812", + "source": "arxiv", + "source_id": "arxiv:2606.02812", + "pdf_url": "https://arxiv.org/pdf/2606.02812", + "primary_query": "agent-memory" + }, + { + "id": "2606.01528", + "title": "Joint Agent Memory and Exploration Learning via Novelty Signals", + "url": "https://arxiv.org/abs/2606.01528", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Shizuo Tian", + "Xiaohong Weng", + "Rui Kong", + "Yuxuan Chen", + "Guohong Liu", + "Yuebing Song", + "Jiacheng Liu", + "Yuchen Li", + "Dawei Yin", + "Ting Cao", + "Yunxin Liu", + "Yuanchun Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.01528", + "source": "arxiv", + "source_id": "arxiv:2606.01528", + "pdf_url": "https://arxiv.org/pdf/2606.01528", + "primary_query": "agent-memory" + }, + { + "id": "2606.01041", + "title": "ExpWeaver: LLM Agents Learn from Experience via Latent RAG", + "url": "https://arxiv.org/abs/2606.01041", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Tao Feng", + "Tianyang Luo", + "Jingjun Xu", + "Zhigang Hua", + "Yan Xie", + "Shuang Yang", + "Ge Liu", + "Jiaxuan You" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent", + "rag-agent" + ], + "arxiv_id": "2606.01041", + "source": "arxiv", + "source_id": "arxiv:2606.01041", + "pdf_url": "https://arxiv.org/pdf/2606.01041", + "primary_query": "planning-agent" + }, + { + "id": "2605.30907", + "title": "BlueFin: Benchmarking LLM Agents on Financial Spreadsheets", + "url": "https://arxiv.org/abs/2605.30907", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Srivatsa Kundurthy", + "Clara Na", + "Colton Moraine", + "Anoushka Mohta", + "Case Winter", + "George Fang", + "John Ling", + "Emma Strubell", + "Zach Kirshner" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.30907", + "source": "arxiv", + "source_id": "arxiv:2605.30907", + "pdf_url": "https://arxiv.org/pdf/2605.30907", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.30858", + "title": "ForecastCompass: Guiding Agentic Forecasting with Adaptive Factor Memory", + "url": "https://arxiv.org/abs/2605.30858", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Yurui Chang", + "Yongkang Du", + "Yuanpu Cao", + "Jinghui Chen", + "Lu Lin" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.30858", + "source": "arxiv", + "source_id": "arxiv:2605.30858", + "pdf_url": "https://arxiv.org/pdf/2605.30858", + "primary_query": "agent-memory" + }, + { + "id": "2605.30058", + "title": "HEART-Bench: Do LLM Agents Exhibit Human-like Psychology?", + "url": "https://arxiv.org/abs/2605.30058", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Weihan Peng", + "Chenxu Zhang", + "Qianao Wang", + "Yuling Shi", + "Heng Lian", + "Qihong Mao", + "Jiahao Pang", + "Chunliang Feng", + "Bowen Li", + "Xiaodong Gu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.30058", + "source": "arxiv", + "source_id": "arxiv:2605.30058", + "pdf_url": "https://arxiv.org/pdf/2605.30058", + "primary_query": "planning-agent" + }, + { + "id": "2605.27762", + "title": "PEAM: Parametric Embodied Agent Memory through Contrastive Internalization of Experience in Minecraft", + "url": "https://arxiv.org/abs/2605.27762", + "published": "2026-05-26", + "updated": "2026-06-01", + "authors": [ + "Yuchen Guo", + "Junli Gong", + "Weicheng Wang", + "Hongmin Cai", + "Yiu-ming Cheung", + "Weifeng Su" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.27762", + "source": "arxiv", + "source_id": "arxiv:2605.27762", + "pdf_url": "https://arxiv.org/pdf/2605.27762", + "primary_query": "agent-memory" + }, + { + "id": "2605.24659", + "title": "IterInject: Indirect Prompt Injection Against LLM Agents via Feedback-Guided Iterative Optimization", + "url": "https://arxiv.org/abs/2605.24659", + "published": "2026-05-23", + "updated": "2026-05-23", + "authors": [ + "Zixuan Chen", + "Jiaxiang Chen", + "Li Luo", + "Ke Xu", + "Xiaoxiang Huang", + "Tanfeng Sun", + "Xinghao Jiang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.24659", + "source": "arxiv", + "source_id": "arxiv:2605.24659", + "pdf_url": "https://arxiv.org/pdf/2605.24659", + "primary_query": "planning-agent" + }, + { + "id": "2605.23723", + "title": "MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection", + "url": "https://arxiv.org/abs/2605.23723", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Zhewen Tan", + "Yilun Yao", + "Huiyan Jin", + "Wenhan Yu", + "Guoan Wang", + "Mengyuan Fan", + "liang lu", + "Feng Liu", + "Xiangzheng Zhang", + "Duohe Ma", + "Tong Yang", + "Lin Sun" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.23723", + "source": "arxiv", + "source_id": "arxiv:2605.23723", + "pdf_url": "https://arxiv.org/pdf/2605.23723", + "primary_query": "agent-memory" + }, + { + "id": "2605.24219", + "title": "Beyond Final Answers: Auditing Trajectory-Level Hallucinations in Multi-Agent Industrial Workflows", + "url": "https://arxiv.org/abs/2605.24219", + "published": "2026-05-22", + "updated": "2026-05-26", + "authors": [ + "Harshada Badave", + "Santosh Borse", + "Andrea Gomez", + "Harshitha Narahari", + "Sara Carter", + "Vishwa Bhatt", + "Aishani Rachakonda", + "Shuxin Lin", + "Dhaval Patel" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.24219", + "source": "arxiv", + "source_id": "arxiv:2605.24219", + "pdf_url": "https://arxiv.org/pdf/2605.24219", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.22154", + "title": "IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents", + "url": "https://arxiv.org/abs/2605.22154", + "published": "2026-05-21", + "updated": "2026-05-21", + "authors": [ + "Daewon Choi", + "Kyunghyun Park", + "Woomin Song", + "Saket Dingliwal", + "Sai Muralidhar Jayanthi", + "Jinwoo Shin", + "Aram Galstyan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.22154", + "source": "arxiv", + "source_id": "arxiv:2605.22154", + "pdf_url": "https://arxiv.org/pdf/2605.22154", + "primary_query": "planning-agent" + }, + { + "id": "2605.21240", + "title": "APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents", + "url": "https://arxiv.org/abs/2605.21240", + "published": "2026-05-20", + "updated": "2026-05-20", + "authors": [ + "Yibo Li", + "Jiashuo Yang", + "Zhi Zheng", + "Zhiyuan Hu", + "Yuan Sui", + "Shizun Wang", + "Yufei He", + "Bryan Hooi" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.21240", + "source": "arxiv", + "source_id": "arxiv:2605.21240", + "pdf_url": "https://arxiv.org/pdf/2605.21240", + "primary_query": "planning-agent" + }, + { + "id": "2605.18930", + "title": "OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences", + "url": "https://arxiv.org/abs/2605.18930", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Kaixiang Wang", + "Jiong Lou", + "Zhaojiacheng Zhou", + "Jie Li" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.18930", + "source": "arxiv", + "source_id": "arxiv:2605.18930", + "pdf_url": "https://arxiv.org/pdf/2605.18930", + "primary_query": "agent-memory" + }, + { + "id": "2605.18284", + "title": "CommitDistill: A Lightweight Knowledge-Centric Memory Layer for Software Repositories", + "url": "https://arxiv.org/abs/2605.18284", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Divya Chukkapalli", + "Thejesh Avula", + "Aditya Aggarwal", + "Harsimran Singh", + "Amith Tallanki" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.18284", + "source": "arxiv", + "source_id": "arxiv:2605.18284", + "pdf_url": "https://arxiv.org/pdf/2605.18284", + "primary_query": "agent-memory" + }, + { + "id": "2605.14421", + "title": "MemLineage: Lineage-Guided Enforcement for LLM Agent Memory", + "url": "https://arxiv.org/abs/2605.14421", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Ciyan Ouyang", + "Rui Hou" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.14421", + "source": "arxiv", + "source_id": "arxiv:2605.14421", + "pdf_url": "https://arxiv.org/pdf/2605.14421", + "primary_query": "agent-memory" + }, + { + "id": "2605.14892", + "title": "Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems", + "url": "https://arxiv.org/abs/2605.14892", + "published": "2026-05-14", + "updated": "2026-05-15", + "authors": [ + "Shihao Qi", + "Jie Ma", + "Rui Xing", + "Wei Guo", + "Xiao Huang", + "Zhitao Gao", + "Jianhao Deng", + "Jun Liu", + "Lingling Zhang", + "Bifan Wei", + "Boqian Yang", + "Pinghui Wang", + "Jianwen Sun", + "Jing Tao", + "Yaqiang Wu", + "Hui Liu", + "Yu Yao", + "Tongliang Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14892", + "source": "arxiv", + "source_id": "arxiv:2605.14892", + "pdf_url": "https://arxiv.org/pdf/2605.14892", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.14527", + "title": "Lang2MLIP: End-to-End Language-to-Machine Learning Interatomic Potential Development with Autonomous Agentic Workflows", + "url": "https://arxiv.org/abs/2605.14527", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Wenwen Li", + "Yuki Orimo", + "Nontawat Charoenphakdee" + ], + "categories": [ + "cs.LG", + "cond-mat.mtrl-sci", + "physics.comp-ph" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14527", + "source": "arxiv", + "source_id": "arxiv:2605.14527", + "pdf_url": "https://arxiv.org/pdf/2605.14527", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.13481", + "title": "PersonalAI 2.0: Enhancing knowledge graph traversal/retrieval with planning mechanism for Personalized LLM Agents", + "url": "https://arxiv.org/abs/2605.13481", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Mikhail Menschikov", + "Matvey Iskornev", + "Alexander Kharitonov", + "Alina Bogdanova", + "Mikhail Belkin", + "Ekaterina Lisitsyna", + "Artyom Sosedka", + "Victoria Dochkina", + "Ruslan Kostoev", + "Ilia Perepechkin", + "Evgeny Burnaev" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.13481", + "source": "arxiv", + "source_id": "arxiv:2605.13481", + "pdf_url": "https://arxiv.org/pdf/2605.13481", + "primary_query": "planning-agent" + }, + { + "id": "2605.12260", + "title": "PRISM: Pareto-Efficient Retrieval over Intent-Aware Structured Memory for Long-Horizon Agents", + "url": "https://arxiv.org/abs/2605.12260", + "published": "2026-05-12", + "updated": "2026-05-22", + "authors": [ + "Jingyi Peng", + "Zhongwei Wan", + "Weiting Liu", + "Qiuzhuang Sun" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.12260", + "source": "arxiv", + "source_id": "arxiv:2605.12260", + "pdf_url": "https://arxiv.org/pdf/2605.12260", + "primary_query": "language-agent" + }, + { + "id": "2605.11225", + "title": "PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement", + "url": "https://arxiv.org/abs/2605.11225", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Tuo Zhang", + "Alin-Ionut Popa", + "Yan Xu", + "Rui Song", + "Dimitrios Dimitriadis" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2605.11225", + "source": "arxiv", + "source_id": "arxiv:2605.11225", + "pdf_url": "https://arxiv.org/pdf/2605.11225", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.09692", + "title": "Causal state binding predicts action control in language agents", + "url": "https://arxiv.org/abs/2605.09692", + "published": "2026-05-10", + "updated": "2026-06-01", + "authors": [ + "Xiao Jia" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.09692", + "source": "arxiv", + "source_id": "arxiv:2605.09692", + "pdf_url": "https://arxiv.org/pdf/2605.09692", + "primary_query": "language-agent" + }, + { + "id": "2605.07251", + "title": "Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning", + "url": "https://arxiv.org/abs/2605.07251", + "published": "2026-05-08", + "updated": "2026-05-08", + "authors": [ + "Yuyang Wu", + "Yue Huang", + "Shuaike Shen", + "Xujian Wang", + "Shuhao Zhang", + "Qiyao Xue", + "Weichen Liu", + "Runtian Gao", + "Jian Ma", + "Xiangliang Zhang", + "Olexandr Isayev" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.07251", + "source": "arxiv", + "source_id": "arxiv:2605.07251", + "pdf_url": "https://arxiv.org/pdf/2605.07251", + "primary_query": "planning-agent" + }, + { + "id": "2605.06713", + "title": "Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand", + "url": "https://arxiv.org/abs/2605.06713", + "published": "2026-05-06", + "updated": "2026-05-06", + "authors": [ + "Christopher Koch" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-safety", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "planning-agent" + ], + "arxiv_id": "2605.06713", + "source": "arxiv", + "source_id": "arxiv:2605.06713", + "pdf_url": "https://arxiv.org/pdf/2605.06713", + "primary_query": "agent-safety" + }, + { + "id": "2605.02240", + "title": "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments", + "url": "https://arxiv.org/abs/2605.02240", + "published": "2026-05-04", + "updated": "2026-05-04", + "authors": [ + "Ruoqi Liu", + "Imran Q. Mohiuddin", + "Austin J. Schoeffler", + "Kavita Renduchintala", + "Ashwin Nayak", + "Prasantha L. Vemu", + "Shivam C. Vedak", + "Kameron C. Black", + "John L. Havlik", + "Isaac Ogunmola", + "Stephen P. Ma", + "Roopa Dhatt", + "Jonathan H. Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.02240", + "source": "arxiv", + "source_id": "arxiv:2605.02240", + "pdf_url": "https://arxiv.org/pdf/2605.02240", + "primary_query": "planning-agent" + }, + { + "id": "2604.11557", + "title": "UniToolCall: Unifying Tool-Use Representation, Data, and Evaluation for LLM Agents", + "url": "https://arxiv.org/abs/2604.11557", + "published": "2026-04-13", + "updated": "2026-05-25", + "authors": [ + "Yijuan Liang", + "Xinghao Chen", + "Yifan Ge", + "Ziyi Wu", + "Hao Wu", + "Changyu Zeng", + "Wei Xing", + "Xiaoyu Shen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2604.11557", + "source": "arxiv", + "source_id": "arxiv:2604.11557", + "pdf_url": "https://arxiv.org/pdf/2604.11557", + "primary_query": "function-calling" + }, + { + "id": "2604.10577", + "title": "The Blind Spot of Agent Safety: How Benign User Instructions Expose Critical Vulnerabilities in Computer-Use Agents", + "url": "https://arxiv.org/abs/2604.10577", + "published": "2026-04-12", + "updated": "2026-04-17", + "authors": [ + "Xuwei Ding", + "Skylar Zhai", + "Linxin Song", + "Jiate Li", + "Taiwei Shi", + "Nicholas Meade", + "Siva Reddy", + "Jian Kang", + "Jieyu Zhao" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.10577", + "source": "arxiv", + "source_id": "arxiv:2604.10577", + "pdf_url": "https://arxiv.org/pdf/2604.10577", + "primary_query": "agent-safety" + }, + { + "id": "2603.24257", + "title": "Memory-Augmented Vision-Language Agents for Persistent and Semantically Consistent Object Captioning", + "url": "https://arxiv.org/abs/2603.24257", + "published": "2026-03-25", + "updated": "2026-03-30", + "authors": [ + "Tommaso Galliena", + "Stefano Rosa", + "Tommaso Apicella", + "Pietro Morerio", + "Alessio Del Bue", + "Lorenzo Natale" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.24257", + "source": "arxiv", + "source_id": "arxiv:2603.24257", + "pdf_url": "https://arxiv.org/pdf/2603.24257", + "primary_query": "language-agent" + }, + { + "id": "2603.10492", + "title": "Human-AI Co-reasoning for Clinical Diagnosis with Evidence-Integrated Language Agent", + "url": "https://arxiv.org/abs/2603.10492", + "published": "2026-03-11", + "updated": "2026-03-18", + "authors": [ + "Zhongzhen Huang", + "Yan Ling", + "Hong Chen", + "Ye Feng", + "Li Wu", + "Linjie Mu", + "Shaoting Zhang", + "Xiaofan Zhang", + "Kun Qian", + "Xiaomu Li" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.10492", + "source": "arxiv", + "source_id": "arxiv:2603.10492", + "pdf_url": "https://arxiv.org/pdf/2603.10492", + "primary_query": "language-agent" + }, + { + "id": "2603.07496", + "title": "From Thinker to Society: Security in Hierarchical Autonomy Evolution of AI Agents", + "url": "https://arxiv.org/abs/2603.07496", + "published": "2026-03-08", + "updated": "2026-03-21", + "authors": [ + "Xiaolei Zhang", + "Lu Zhou", + "Xiaogang Xu", + "Jiafei Wu", + "Tianyu Du", + "Heqing Huang", + "Hao Peng", + "Zhe Liu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.07496", + "source": "arxiv", + "source_id": "arxiv:2603.07496", + "pdf_url": "https://arxiv.org/pdf/2603.07496", + "primary_query": "agent-safety" + }, + { + "id": "2603.03680", + "title": "MAGE: Meta-Reinforcement Learning for Language Agents toward Strategic Exploration and Exploitation", + "url": "https://arxiv.org/abs/2603.03680", + "published": "2026-03-04", + "updated": "2026-03-04", + "authors": [ + "Lu Yang", + "Zelai Xu", + "Minyang Xie", + "Jiaxuan Gao", + "Zhao Shok", + "Yu Wang", + "Yi Wu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.03680", + "source": "arxiv", + "source_id": "arxiv:2603.03680", + "pdf_url": "https://arxiv.org/pdf/2603.03680", + "primary_query": "language-agent" + }, + { + "id": "2603.02711", + "title": "A Natural Language Agentic Approach to Study Affective Polarization", + "url": "https://arxiv.org/abs/2603.02711", + "published": "2026-03-03", + "updated": "2026-03-03", + "authors": [ + "Stephanie Anneris Malvicini", + "Ewelina Gajewska", + "Arda Derbent", + "Katarzyna Budzynska", + "Jarosław A. Chudziak", + "Maria Vanina Martinez" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "multi-agent", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.02711", + "source": "arxiv", + "source_id": "arxiv:2603.02711", + "pdf_url": "https://arxiv.org/pdf/2603.02711", + "primary_query": "language-agent" + }, + { + "id": "2603.03515", + "title": "The Controllability Trap: A Governance Framework for Military AI Agents", + "url": "https://arxiv.org/abs/2603.03515", + "published": "2026-03-03", + "updated": "2026-03-03", + "authors": [ + "Subramanyam Sahoo" + ], + "categories": [ + "cs.CY", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "world-model" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.03515", + "source": "arxiv", + "source_id": "arxiv:2603.03515", + "pdf_url": "https://arxiv.org/pdf/2603.03515", + "primary_query": "agent-safety" + }, + { + "id": "2604.03242", + "title": "DRAFT: Task Decoupled Latent Reasoning for Agent Safety", + "url": "https://arxiv.org/abs/2604.03242", + "published": "2026-02-11", + "updated": "2026-02-11", + "authors": [ + "Lin Wang", + "Junfeng Fang", + "Dan Zhang", + "Fei Shen", + "Xiang Wang", + "Tat-Seng Chua" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.03242", + "source": "arxiv", + "source_id": "arxiv:2604.03242", + "pdf_url": "https://arxiv.org/pdf/2604.03242", + "primary_query": "agent-safety" + }, + { + "id": "2603.08721", + "title": "KernelCraft: Benchmarking for Agentic Close-to-Metal Kernel Generation on Emerging Hardware", + "url": "https://arxiv.org/abs/2603.08721", + "published": "2026-02-10", + "updated": "2026-05-29", + "authors": [ + "Jiayi Nie", + "Haoran Wu", + "Yao Lai", + "Zeyu Cao", + "Cheng Zhang", + "Binglei Lou", + "Erwei Wang", + "Jianyi Cheng", + "Timothy M. Jones", + "Robert Mullins", + "Rika Antonova", + "Yiren Zhao" + ], + "categories": [ + "cs.AR", + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "workflow-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.08721", + "source": "arxiv", + "source_id": "arxiv:2603.08721", + "pdf_url": "https://arxiv.org/pdf/2603.08721", + "primary_query": "function-calling" + }, + { + "id": "2602.03224", + "title": "TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking", + "url": "https://arxiv.org/abs/2602.03224", + "published": "2026-02-03", + "updated": "2026-06-06", + "authors": [ + "Yu Cheng", + "Yongkang Hu", + "Jiuan Zhou", + "Yushuo Zhang", + "Yihang Chen", + "Huichi Zhou", + "Mingang Chen", + "Zhizhong Zhang", + "Kun Shao", + "Yuan Xie", + "Zhaoxia Yin" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "reasoning" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.03224", + "source": "arxiv", + "source_id": "arxiv:2602.03224", + "pdf_url": "https://arxiv.org/pdf/2602.03224", + "primary_query": "agent-safety" + }, + { + "id": "2512.11682", + "title": "MedAI: Evaluating TxAgent's Therapeutic Agentic Reasoning in the NeurIPS CURE-Bench Competition", + "url": "https://arxiv.org/abs/2512.11682", + "published": "2025-12-12", + "updated": "2026-06-15", + "authors": [ + "Tim Cofala", + "Christian Kalfar", + "Jingge Xiao", + "Johanna Schrader", + "Michelle Tang", + "Wolfgang Nejdl" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.11682", + "source": "arxiv", + "source_id": "arxiv:2512.11682", + "pdf_url": "https://arxiv.org/pdf/2512.11682", + "primary_query": "function-calling" + }, + { + "id": "2511.15203", + "title": "Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks", + "url": "https://arxiv.org/abs/2511.15203", + "published": "2025-11-19", + "updated": "2025-11-19", + "authors": [ + "Zimo Ji", + "Xunguang Wang", + "Zongjie Li", + "Pingchuan Ma", + "Yudong Gao", + "Daoyuan Wu", + "Xincheng Yan", + "Tian Tian", + "Shuai Wang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2511.15203", + "source": "arxiv", + "source_id": "arxiv:2511.15203", + "pdf_url": "https://arxiv.org/pdf/2511.15203", + "primary_query": "function-calling" + }, + { + "id": "2510.18586", + "title": "TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications", + "url": "https://arxiv.org/abs/2510.18586", + "published": "2025-10-21", + "updated": "2026-05-20", + "authors": [ + "Zhuohang Bian", + "Feiyang Wu", + "Zhuoran Li", + "Teng Ma", + "Youwei Zhuo" + ], + "categories": [ + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent" + ], + "score": 17, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.18586", + "source": "arxiv", + "source_id": "arxiv:2510.18586", + "pdf_url": "https://arxiv.org/pdf/2510.18586", + "primary_query": "function-calling" + }, + { + "id": "2607.06001", + "title": "Information Limits and Attractor Dynamics in Economies of Frontier LLM Agents: A Pre-Registered Test", + "url": "https://arxiv.org/abs/2607.06001", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Cheng Qian" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.06001", + "source": "arxiv", + "source_id": "arxiv:2607.06001", + "pdf_url": "https://arxiv.org/pdf/2607.06001", + "primary_query": "llm-agent" + }, + { + "id": "2607.06000", + "title": "Context-to-Execution Integrity for LLM Agents", + "url": "https://arxiv.org/abs/2607.06000", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Igor Santos-Grueiro" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.06000", + "source": "arxiv", + "source_id": "arxiv:2607.06000", + "pdf_url": "https://arxiv.org/pdf/2607.06000", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.06413", + "title": "An Experimental Design Approach to Evaluating Agentic AI's Autonomous Model Discovery", + "url": "https://arxiv.org/abs/2607.06413", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Hao He", + "Xueying Liu", + "Chris J. Kuhlman", + "Xinwei Deng" + ], + "categories": [ + "stat.ME", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2607.06413", + "source": "arxiv", + "source_id": "arxiv:2607.06413", + "pdf_url": "https://arxiv.org/pdf/2607.06413", + "primary_query": "agentic-ai" + }, + { + "id": "2607.06411", + "title": "RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications", + "url": "https://arxiv.org/abs/2607.06411", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Evgeny Shilov" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2607.06411", + "source": "arxiv", + "source_id": "arxiv:2607.06411", + "pdf_url": "https://arxiv.org/pdf/2607.06411", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.06101", + "title": "Agents That Teach: Towards Designing Incidental Learning Back into AI-Assisted Software Development", + "url": "https://arxiv.org/abs/2607.06101", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Rohit Mehra", + "Samdyuti Suri", + "Prithviraj K Tagadinamani", + "Kapil Singi", + "Vikrant Kaulgud", + "Adam P. Burden" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CY", + "cs.HC" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.06101", + "source": "arxiv", + "source_id": "arxiv:2607.06101", + "pdf_url": "https://arxiv.org/pdf/2607.06101", + "primary_query": "coding-agent" + }, + { + "id": "2607.05659", + "title": "Agents with Feelings? Personality and Emotion in Multi-Agent Software Teams", + "url": "https://arxiv.org/abs/2607.05659", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Yunyan Ding", + "Thomas Zimmermann", + "Iftekhar Ahmed" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.05659", + "source": "arxiv", + "source_id": "arxiv:2607.05659", + "pdf_url": "https://arxiv.org/pdf/2607.05659", + "primary_query": "llm-agent" + }, + { + "id": "2607.04713", + "title": "RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents", + "url": "https://arxiv.org/abs/2607.04713", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Qiang Liu", + "Taian Guo", + "Ruizhi Qiao", + "Xing Sun" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "llm-agent" + ], + "arxiv_id": "2607.04713", + "source": "arxiv", + "source_id": "arxiv:2607.04713", + "pdf_url": "https://arxiv.org/pdf/2607.04713", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.04569", + "title": "LLMs for Agentic Home Energy Management", + "url": "https://arxiv.org/abs/2607.04569", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Sokipriala Jonah" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling", + "llm-agent" + ], + "arxiv_id": "2607.04569", + "source": "arxiv", + "source_id": "arxiv:2607.04569", + "pdf_url": "https://arxiv.org/pdf/2607.04569", + "primary_query": "function-calling" + }, + { + "id": "2607.04617", + "title": "MRMS: A Multi-Resolution Memory Substrate for Long-Lived AI Agents", + "url": "https://arxiv.org/abs/2607.04617", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Jizhizi Li", + "Amy Shi-Nash" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.04617", + "source": "arxiv", + "source_id": "arxiv:2607.04617", + "pdf_url": "https://arxiv.org/pdf/2607.04617", + "primary_query": "ai-agent" + }, + { + "id": "2607.05363", + "title": "SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints", + "url": "https://arxiv.org/abs/2607.05363", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Dylan Zongmin Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "memory", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "tool-use" + ], + "arxiv_id": "2607.05363", + "source": "arxiv", + "source_id": "arxiv:2607.05363", + "pdf_url": "https://arxiv.org/pdf/2607.05363", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.05458", + "title": "Learning to Control LLM Agent Harnesses with Offline Reinforcement Learning", + "url": "https://arxiv.org/abs/2607.05458", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Haiwen Yi", + "Xinyuan Song" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.05458", + "source": "arxiv", + "source_id": "arxiv:2607.05458", + "pdf_url": "https://arxiv.org/pdf/2607.05458", + "primary_query": "llm-agent" + }, + { + "id": "2607.04219", + "title": "Agentic IoT: Architectures, Applications, and Challenges Toward the Internet of Agents", + "url": "https://arxiv.org/abs/2607.04219", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Rümeysa Hilal Sevinç", + "Bahaeddin Türkoğlu", + "İbrahim Kök" + ], + "categories": [ + "cs.AI", + "cs.MA", + "cs.NI" + ], + "topics": [ + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "tool-use" + ], + "arxiv_id": "2607.04219", + "source": "arxiv", + "source_id": "arxiv:2607.04219", + "pdf_url": "https://arxiv.org/pdf/2607.04219", + "primary_query": "ai-agent" + }, + { + "id": "2607.04212", + "title": "An Evaluation of Role-Based Multi-Agent Code Generation on Repository-Scale Problems", + "url": "https://arxiv.org/abs/2607.04212", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Benedetta Donato", + "Noah Hagar-Dent", + "Aaron Worsnop", + "Leonardo Mariani", + "Valerio Terragni" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.04212", + "source": "arxiv", + "source_id": "arxiv:2607.04212", + "pdf_url": "https://arxiv.org/pdf/2607.04212", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.04149", + "title": "Beyond Scene Priors: Fine-Grained Traffic Scene Reasoning with Benchmarking and Query-Guided Small-Object Focus", + "url": "https://arxiv.org/abs/2607.04149", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Waikit Xiu", + "Qiang Lu", + "Zian Wang", + "Xinjie Yang", + "Zhiwei Chen", + "Chen Sun", + "Xiying Li" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.04149", + "source": "arxiv", + "source_id": "arxiv:2607.04149", + "pdf_url": "https://arxiv.org/pdf/2607.04149", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.04034", + "title": "The \"I Don't Know\" Filter: Enhancing Agentic Reliability in Function Calling", + "url": "https://arxiv.org/abs/2607.04034", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Stefan Broecker", + "Mason del Rosario", + "Boris Selitser", + "Thomas Strohmer" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "function-calling" + ], + "arxiv_id": "2607.04034", + "source": "arxiv", + "source_id": "arxiv:2607.04034", + "pdf_url": "https://arxiv.org/pdf/2607.04034", + "primary_query": "agent-evaluation" + }, + { + "id": "2607.03691", + "title": "Don't Blame the Large Language Model: How Scaffolding Evolution Shapes Coding Agent Quality", + "url": "https://arxiv.org/abs/2607.03691", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Oussama Ben Sghaier", + "Hao Li", + "Bram Adams", + "Ahmed E. Hassan" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03691", + "source": "arxiv", + "source_id": "arxiv:2607.03691", + "pdf_url": "https://arxiv.org/pdf/2607.03691", + "primary_query": "coding-agent" + }, + { + "id": "2607.02927", + "title": "VideoSearcher: Empowering Video Deep Research with Multi-Tool Agentic Reasoning via Reinforcement Learning", + "url": "https://arxiv.org/abs/2607.02927", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Zhenkun Gao", + "Yicheng Bao", + "Jinlong Peng", + "Xueheng Li", + "Theo Huang", + "Bangwei Liu", + "Kunquan Li", + "Zhenye Gan", + "Tao Hu", + "Chengjun Xie", + "Mingqian Yang", + "Xuanhua He", + "Zhizhong Zhang", + "Xin Tan", + "Chengjie Wang", + "Yuan Xie" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.02927", + "source": "arxiv", + "source_id": "arxiv:2607.02927", + "pdf_url": "https://arxiv.org/pdf/2607.02927", + "primary_query": "tool-use" + }, + { + "id": "2607.03525", + "title": "GameEngineBench: Evaluating Coding Agents on Real C++ Runtime Environments", + "url": "https://arxiv.org/abs/2607.03525", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Brian La", + "Sejoon Chang", + "Ben Kim", + "Junyoung Bae", + "Aamish Ahmad Beg", + "Sei Chang", + "Gonzalo Gonzalez-Pumariega" + ], + "categories": [ + "cs.SE", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "tool-use", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03525", + "source": "arxiv", + "source_id": "arxiv:2607.03525", + "pdf_url": "https://arxiv.org/pdf/2607.03525", + "primary_query": "coding-agent" + }, + { + "id": "2607.02689", + "title": "S-EMBER: A Large-Scale Benchmark for Streaming Egocentric Memory Retrieval", + "url": "https://arxiv.org/abs/2607.02689", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Xiaodong Wang", + "Xuanyi Zhao", + "Pedro Rodriguez", + "Devendra Singh Sachan", + "Barlas Oguz", + "Seungwhan Moon", + "Shang-Wen Li", + "Gargi Ghosh", + "Xin Dong", + "Wen-Tau Yih" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.02689", + "source": "arxiv", + "source_id": "arxiv:2607.02689", + "pdf_url": "https://arxiv.org/pdf/2607.02689", + "primary_query": "ai-agent" + }, + { + "id": "2607.01788", + "title": "KRCA: An Efficient Root Cause Analysis System in Hyper-Scale Microservice Systems via Agentic AI", + "url": "https://arxiv.org/abs/2607.01788", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Jiamin Jiang", + "Jingfei Feng", + "Yu Luo", + "Qingliang Zhang", + "Yongqian Su", + "Wenwei Gu", + "Shenglin Zhang", + "Tianyu Cui", + "Yao Wu", + "Jielong Huang", + "Nan Qi", + "Dan Pei" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.01788", + "source": "arxiv", + "source_id": "arxiv:2607.01788", + "pdf_url": "https://arxiv.org/pdf/2607.01788", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02716", + "title": "Evaluating Large Language Models for Decision-Making in Agent-Based Urban Mobility Simulations", + "url": "https://arxiv.org/abs/2607.02716", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Bruno Cascaes Alves", + "Míriam Blank Born", + "Ulisses Gilioli Francescatto Júnior", + "Felipe Moura Goulart", + "Letícia Brandão Caldas", + "Marilton Sanchotene de Aguiar" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.02716", + "source": "arxiv", + "source_id": "arxiv:2607.02716", + "pdf_url": "https://arxiv.org/pdf/2607.02716", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.02186", + "title": "UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development", + "url": "https://arxiv.org/abs/2607.02186", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Temitayo Olamilekan Ogunsusi", + "Lijun Qian", + "Xishuang Dong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.02186", + "source": "arxiv", + "source_id": "arxiv:2607.02186", + "pdf_url": "https://arxiv.org/pdf/2607.02186", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.01084", + "title": "Can Agents Generalize to the Open World? Unveiling the Fragility of Static Training in Tool Use", + "url": "https://arxiv.org/abs/2607.01084", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Song-Lin Lv", + "Weiming Wu", + "Rui Zhu", + "Zi-Jian Cheng", + "Lan-Zhe Guo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.01084", + "source": "arxiv", + "source_id": "arxiv:2607.01084", + "pdf_url": "https://arxiv.org/pdf/2607.01084", + "primary_query": "llm-agent" + }, + { + "id": "2607.02599", + "title": "AgentLTL: A Trace-Verification Framework for Measuring, Enforcing, and Training Procedural Compliance in Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2607.02599", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Laïla Elkoussy", + "Julien Perez" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.LO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.02599", + "source": "arxiv", + "source_id": "arxiv:2607.02599", + "pdf_url": "https://arxiv.org/pdf/2607.02599", + "primary_query": "llm-agent" + }, + { + "id": "2607.00555", + "title": "Rise From The Ashes: LLM-based Static Analysis for Deep Learning Framework Bugs", + "url": "https://arxiv.org/abs/2607.00555", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Shaoyu Yang", + "Haifeng Lin", + "Chunrong Fang", + "Xiang Chen", + "Wei Cheng", + "Jiawei Liu", + "Yiyu Zhang", + "Hongyu Liu", + "Zhenyu Chen" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2607.00555", + "source": "arxiv", + "source_id": "arxiv:2607.00555", + "pdf_url": "https://arxiv.org/pdf/2607.00555", + "primary_query": "agentic-ai" + }, + { + "id": "2607.00436", + "title": "PHREEQC-MCQ-200: A Diagnostic Benchmark for Tool-Augmented Scientific Simulator Agents", + "url": "https://arxiv.org/abs/2607.00436", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Ke Zhang", + "Sahchit Chundur", + "Mohammad Javad Qomi", + "Maziar Raissi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.00436", + "source": "arxiv", + "source_id": "arxiv:2607.00436", + "pdf_url": "https://arxiv.org/pdf/2607.00436", + "primary_query": "tool-use" + }, + { + "id": "2607.00502", + "title": "A Task-State Representation for Long-Horizon Mobile GUI Agents", + "url": "https://arxiv.org/abs/2607.00502", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Yujie Zheng", + "Zikang Liu", + "Xin Zhao", + "Ji-Rong Wen" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.00502", + "source": "arxiv", + "source_id": "arxiv:2607.00502", + "pdf_url": "https://arxiv.org/pdf/2607.00502", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.00440", + "title": "Minos: A Multi-Agent Collaborative Framework for Provenance-Based Backward Tracking", + "url": "https://arxiv.org/abs/2607.00440", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Jiahui Wang", + "Zhenyuan Li", + "Zhengkai Wang", + "Xiangmin Shen", + "Fan Zhang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00440", + "source": "arxiv", + "source_id": "arxiv:2607.00440", + "pdf_url": "https://arxiv.org/pdf/2607.00440", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00233", + "title": "From Signals to Structure: How Memory Architecture Drives Language Emergence in LLM Agents", + "url": "https://arxiv.org/abs/2607.00233", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yashar Talebirad", + "Eden Redman", + "Ali Parsaee", + "Osmar R. Zaiane" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.IT", + "cs.MA" + ], + "topics": [ + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00233", + "source": "arxiv", + "source_id": "arxiv:2607.00233", + "pdf_url": "https://arxiv.org/pdf/2607.00233", + "primary_query": "llm-agent" + }, + { + "id": "2606.32034", + "title": "QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2606.32034", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Sergio Hernández-Gutiérrez", + "Matteo Merler", + "Ilze Amanda Auzina", + "Joschka Strüber", + "Ameya Prabhu", + "Matthias Bethge" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.32034", + "source": "arxiv", + "source_id": "arxiv:2606.32034", + "pdf_url": "https://arxiv.org/pdf/2606.32034", + "primary_query": "llm-agent" + }, + { + "id": "2607.02579", + "title": "When Not to Write Memory: Governing False Promotion from Correlated Agent Traces", + "url": "https://arxiv.org/abs/2607.02579", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yijiashun Qi", + "Xiang Xu", + "Yuxuan Li" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "language-agent" + ], + "arxiv_id": "2607.02579", + "source": "arxiv", + "source_id": "arxiv:2607.02579", + "pdf_url": "https://arxiv.org/pdf/2607.02579", + "primary_query": "agent-memory" + }, + { + "id": "2606.31339", + "title": "Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems", + "url": "https://arxiv.org/abs/2606.31339", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Guoqin Tang", + "Qingxuan Jia", + "Yichen Tan", + "Zeyuan Huang", + "Ning Ji", + "Gang Chen" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.31339", + "source": "arxiv", + "source_id": "arxiv:2606.31339", + "pdf_url": "https://arxiv.org/pdf/2606.31339", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31980", + "title": "DigitalCoach: Communication and Grounding Gaps in Human and Agentic Computer Use Coaching", + "url": "https://arxiv.org/abs/2606.31980", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Meng Chen", + "Anya Ji", + "Tsung-Han Wu", + "Tobias Maringgele", + "David M. Chan", + "Alane Suhr", + "Amy Pavel" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.31980", + "source": "arxiv", + "source_id": "arxiv:2606.31980", + "pdf_url": "https://arxiv.org/pdf/2606.31980", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.31134", + "title": "Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics", + "url": "https://arxiv.org/abs/2606.31134", + "published": "2026-06-30", + "updated": "2026-07-01", + "authors": [ + "Arshia Soltani Moakhar", + "Iman Gholami", + "Max Springer", + "Mahdi JafariRaviz", + "MohammadTaghi Hajiaghayi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31134", + "source": "arxiv", + "source_id": "arxiv:2606.31134", + "pdf_url": "https://arxiv.org/pdf/2606.31134", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31200", + "title": "Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping", + "url": "https://arxiv.org/abs/2606.31200", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Tao Chen", + "Lizheng Liu", + "Jiaxu Wang", + "Ziyue Jiang", + "Ruiqi Tian", + "JiGuang Huo", + "Zhongxue Gan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.31200", + "source": "arxiv", + "source_id": "arxiv:2606.31200", + "pdf_url": "https://arxiv.org/pdf/2606.31200", + "primary_query": "rag-agent" + }, + { + "id": "2606.30251", + "title": "TACO: Tool-Augmented Credit Optimization for Agentic Tool Use", + "url": "https://arxiv.org/abs/2606.30251", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Mingkuan Feng", + "Jinyang Wu", + "Hao Gu", + "Fangrui Lv", + "Ruihan Jin", + "Chuyuan Zhang", + "Zhengqi Wen", + "Jianhua Tao" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.30251", + "source": "arxiv", + "source_id": "arxiv:2606.30251", + "pdf_url": "https://arxiv.org/pdf/2606.30251", + "primary_query": "tool-use" + }, + { + "id": "2606.30294", + "title": "Rehearsed Multi-Agent Live Product Demonstrations with Real-Time Voice Question Answering", + "url": "https://arxiv.org/abs/2606.30294", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Rahul Khedar", + "Mayank Malhotra", + "Avinash Karn", + "Mouli V", + "Prakhar Mehrotra" + ], + "categories": [ + "cs.AI", + "cs.HC", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.30294", + "source": "arxiv", + "source_id": "arxiv:2606.30294", + "pdf_url": "https://arxiv.org/pdf/2606.30294", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.29354", + "title": "When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning", + "url": "https://arxiv.org/abs/2606.29354", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Zhengqi Pei", + "Qingming Huang", + "Shuhui Wang" + ], + "categories": [ + "cs.AI", + "cs.NE" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.29354", + "source": "arxiv", + "source_id": "arxiv:2606.29354", + "pdf_url": "https://arxiv.org/pdf/2606.29354", + "primary_query": "llm-agent" + }, + { + "id": "2606.29315", + "title": "Hierarchical Experimentalist Agents", + "url": "https://arxiv.org/abs/2606.29315", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Abhranil Chandra", + "Sankaran Vaidyanathan", + "Utsav Dhanuka", + "Varun Gandhi", + "Scott Niekum" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29315", + "source": "arxiv", + "source_id": "arxiv:2606.29315", + "pdf_url": "https://arxiv.org/pdf/2606.29315", + "primary_query": "llm-agent" + }, + { + "id": "2606.29445", + "title": "Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction", + "url": "https://arxiv.org/abs/2606.29445", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Sunqi Fan", + "Qingle Liu", + "Runqi Yin", + "Meng-Hao Guo", + "Shuojin Yang" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.29445", + "source": "arxiv", + "source_id": "arxiv:2606.29445", + "pdf_url": "https://arxiv.org/pdf/2606.29445", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.28841", + "title": "LAMP: Lean-based Agentic framework with MCP and Proof Repair", + "url": "https://arxiv.org/abs/2606.28841", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Santhana Srinivasan R", + "Maithilee Patawar" + ], + "categories": [ + "cs.LO", + "cs.AI", + "cs.CL" + ], + "topics": [ + "coding-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28841", + "source": "arxiv", + "source_id": "arxiv:2606.28841", + "pdf_url": "https://arxiv.org/pdf/2606.28841", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28450", + "title": "LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity", + "url": "https://arxiv.org/abs/2606.28450", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Yiwei Xu", + "Yong Zhuang", + "Xuanming Liu", + "Tian Zhang", + "Bowen Xiao", + "Xiaoyang Xu", + "Delong Jiang", + "Juan Wang", + "Hongxin Hu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.28450", + "source": "arxiv", + "source_id": "arxiv:2606.28450", + "pdf_url": "https://arxiv.org/pdf/2606.28450", + "primary_query": "agent-safety" + }, + { + "id": "2606.28480", + "title": "TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents", + "url": "https://arxiv.org/abs/2606.28480", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Shoufa Chen", + "Luyuan Wang", + "Xuan Yang", + "Zhiheng Liu", + "Yuren Cong", + "Yuanfeng Ji", + "Feiyan Zhou", + "Xiaohui Zhang", + "Fanny Yang", + "Belinda Zeng" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "reasoning", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.28480", + "source": "arxiv", + "source_id": "arxiv:2606.28480", + "pdf_url": "https://arxiv.org/pdf/2606.28480", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.27350", + "title": "CHIA: An open-source framework for principled, agentic AI-driven hardware/software co-design research", + "url": "https://arxiv.org/abs/2606.27350", + "published": "2026-06-25", + "updated": "2026-06-27", + "authors": [ + "Angela Cui", + "Ferran Hermida-Rivera", + "Jack Toubes", + "Raghav Gupta", + "Jim Fang", + "Chengyi Lux Zhang", + "Ella Schwarz", + "Junha Kim", + "Yakun Sophia Shao", + "Borivoje Nikolic", + "Christopher W. Fletcher", + "Sagar Karandikar" + ], + "categories": [ + "cs.AR" + ], + "topics": [ + "agent-safety", + "coding-agent", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2606.27350", + "source": "arxiv", + "source_id": "arxiv:2606.27350", + "pdf_url": "https://arxiv.org/pdf/2606.27350", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26721", + "title": "Knowledge-Based Pull Requests: A Trusted Workflow for Agent-Mediated Knowledge Collaboration", + "url": "https://arxiv.org/abs/2606.26721", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Xinyu Zhang", + "Weiwei Sun" + ], + "categories": [ + "cs.SE", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.26721", + "source": "arxiv", + "source_id": "arxiv:2606.26721", + "pdf_url": "https://arxiv.org/pdf/2606.26721", + "primary_query": "coding-agent" + }, + { + "id": "2606.27330", + "title": "Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning", + "url": "https://arxiv.org/abs/2606.27330", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Tianyi Men", + "Zhuoran Jin", + "Pengfei Cao", + "Yubo Chen", + "Kang Liu", + "Jun Zhao" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.CV", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.27330", + "source": "arxiv", + "source_id": "arxiv:2606.27330", + "pdf_url": "https://arxiv.org/pdf/2606.27330", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.27009", + "title": "Semantic Early-Stopping for Iterative LLM Agent Loops", + "url": "https://arxiv.org/abs/2606.27009", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Sahil Shrivastava" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.27009", + "source": "arxiv", + "source_id": "arxiv:2606.27009", + "pdf_url": "https://arxiv.org/pdf/2606.27009", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.27483", + "title": "Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning", + "url": "https://arxiv.org/abs/2606.27483", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Xuan Zhang", + "Zhijian Zhou", + "Lingfeng Qiao", + "Yulei Qin", + "Ke Li", + "Xing Sun", + "Xiaoyu Tan", + "Chao Qu", + "Yuan Qi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.27483", + "source": "arxiv", + "source_id": "arxiv:2606.27483", + "pdf_url": "https://arxiv.org/pdf/2606.27483", + "primary_query": "planning-agent" + }, + { + "id": "2606.26216", + "title": "CyberChainBench: Can AI Agents Secure Smart Contracts Against Real-World On-Chain Vulnerabilities?", + "url": "https://arxiv.org/abs/2606.26216", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Jintao Huang", + "Fengqing Jiang", + "Radha Poovendran", + "Zhiqiang Lin" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "ai-agent" + ], + "arxiv_id": "2606.26216", + "source": "arxiv", + "source_id": "arxiv:2606.26216", + "pdf_url": "https://arxiv.org/pdf/2606.26216", + "primary_query": "agent-safety" + }, + { + "id": "2606.26203", + "title": "Agentic Analysis for Agentic Infrastructure: An LLM-Powered Pipeline for Comparative Governance of DAO and Corporate AI Protocols", + "url": "https://arxiv.org/abs/2606.26203", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yutian Wang", + "Luyao Zhang" + ], + "categories": [ + "cs.AI", + "cs.CY", + "cs.MA", + "cs.SI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent" + ], + "arxiv_id": "2606.26203", + "source": "arxiv", + "source_id": "arxiv:2606.26203", + "pdf_url": "https://arxiv.org/pdf/2606.26203", + "primary_query": "agentic-ai" + }, + { + "id": "2606.25760", + "title": "Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets", + "url": "https://arxiv.org/abs/2606.25760", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Divake Kumar", + "Sina Tayebati", + "Devashri Naik", + "Amanda Sofie Rios", + "Nilesh Ahuja", + "Omesh Tickoo", + "Ranganath Krishnan", + "Amit Ranjan Trivedi" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "web-gui-agent" + ], + "arxiv_id": "2606.25760", + "source": "arxiv", + "source_id": "arxiv:2606.25760", + "pdf_url": "https://arxiv.org/pdf/2606.25760", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.25651", + "title": "MedGuards: Multi-Agent System for Reliable Medical Error Detection and Correction", + "url": "https://arxiv.org/abs/2606.25651", + "published": "2026-06-24", + "updated": "2026-06-26", + "authors": [ + "Congbo Ma", + "Hu Wang", + "Yichun Zhang", + "Farah E. Shamout" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.25651", + "source": "arxiv", + "source_id": "arxiv:2606.25651", + "pdf_url": "https://arxiv.org/pdf/2606.25651", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24595", + "title": "MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery", + "url": "https://arxiv.org/abs/2606.24595", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Enze Ma", + "Yufan Zhou", + "Wei-Chieh Huang", + "Jie Yang", + "Huanhuan Ma", + "Zixuan Wang", + "Chengze Li", + "Chunyu Miao", + "Philip S. Yu", + "Zhen Wang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.24595", + "source": "arxiv", + "source_id": "arxiv:2606.24595", + "pdf_url": "https://arxiv.org/pdf/2606.24595", + "primary_query": "agent-memory" + }, + { + "id": "2606.24453", + "title": "Bayesian control for coding agents", + "url": "https://arxiv.org/abs/2606.24453", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Theodore Papamarkou", + "Vladislav Smirnov", + "Viktor Mazanov", + "Artem Vazhentsev", + "Preslav Nakov", + "Timothy Baldwin", + "Artem Shelmanov" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "tool-use" + ], + "arxiv_id": "2606.24453", + "source": "arxiv", + "source_id": "arxiv:2606.24453", + "pdf_url": "https://arxiv.org/pdf/2606.24453", + "primary_query": "coding-agent" + }, + { + "id": "2606.24525", + "title": "VisCritic: Visual State Comparison as Process Reward for GUI Agents", + "url": "https://arxiv.org/abs/2606.24525", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Jiachen Qian" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.24525", + "source": "arxiv", + "source_id": "arxiv:2606.24525", + "pdf_url": "https://arxiv.org/pdf/2606.24525", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.24689", + "title": "Automated Summarization of Software Documents: An LLM-based Multi-Agent Approach", + "url": "https://arxiv.org/abs/2606.24689", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Duc S. H. Nguyen", + "Minh T. Nguyen", + "Phuong T. Nguyen", + "Juri Di Rocco", + "Davide Di Ruscio" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24689", + "source": "arxiv", + "source_id": "arxiv:2606.24689", + "pdf_url": "https://arxiv.org/pdf/2606.24689", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24623", + "title": "Privacy-Preserving RAG via Multi-Agent Semantic Rewriting: Achieving Confidentiality Without Compromising Contextual Fidelity", + "url": "https://arxiv.org/abs/2606.24623", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yuanhe Zhao", + "Tianyu Zhang", + "Huafei Xing", + "Derek F. Wong", + "Jianbin Li", + "Tao Fang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "rag-agent" + ], + "arxiv_id": "2606.24623", + "source": "arxiv", + "source_id": "arxiv:2606.24623", + "pdf_url": "https://arxiv.org/pdf/2606.24623", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.23664", + "title": "MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems?", + "url": "https://arxiv.org/abs/2606.23664", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Juyang Bai", + "Laixi Shi" + ], + "categories": [ + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.23664", + "source": "arxiv", + "source_id": "arxiv:2606.23664", + "pdf_url": "https://arxiv.org/pdf/2606.23664", + "primary_query": "agentic-ai" + }, + { + "id": "2606.23032", + "title": "IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO", + "url": "https://arxiv.org/abs/2606.23032", + "published": "2026-06-22", + "updated": "2026-06-30", + "authors": [ + "Mostapha Benhenda" + ], + "categories": [ + "cs.AI", + "q-fin.GN" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.23032", + "source": "arxiv", + "source_id": "arxiv:2606.23032", + "pdf_url": "https://arxiv.org/pdf/2606.23032", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.22741", + "title": "GRADE: Graph Representation of LLM Agent Dependency and Execution", + "url": "https://arxiv.org/abs/2606.22741", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Yue Zhao" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "coding-agent", + "multi-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm", + "tool-use" + ], + "arxiv_id": "2606.22741", + "source": "arxiv", + "source_id": "arxiv:2606.22741", + "pdf_url": "https://arxiv.org/pdf/2606.22741", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.23752", + "title": "ESAA-Conversational: An Event-Sourced Memory Layer for Continuity, Handoff, and Curation Across Heterogeneous LLM Coding Agents", + "url": "https://arxiv.org/abs/2606.23752", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Elzo Brito dos Santos Filho" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "coding-agent", + "memory", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.23752", + "source": "arxiv", + "source_id": "arxiv:2606.23752", + "pdf_url": "https://arxiv.org/pdf/2606.23752", + "primary_query": "coding-agent" + }, + { + "id": "2606.23327", + "title": "VideoAgent: All-in-One Framework for Video Understanding and Editing", + "url": "https://arxiv.org/abs/2606.23327", + "published": "2026-06-22", + "updated": "2026-07-03", + "authors": [ + "Hengji Zhou", + "Lingxuan Huang", + "Jian Wang", + "Bing Zhou", + "Si Wu", + "Lianghao Xia", + "Chao Huang" + ], + "categories": [ + "cs.CV", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.23327", + "source": "arxiv", + "source_id": "arxiv:2606.23327", + "pdf_url": "https://arxiv.org/pdf/2606.23327", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22110", + "title": "TraceView: Interactive Visualization of Agentic Program Repair Trajectories", + "url": "https://arxiv.org/abs/2606.22110", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Amirali Sajadi", + "Tu Nguyen", + "Kimmie Huynh", + "Esteban Parra", + "Preetha Chatterjee" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.22110", + "source": "arxiv", + "source_id": "arxiv:2606.22110", + "pdf_url": "https://arxiv.org/pdf/2606.22110", + "primary_query": "tool-use" + }, + { + "id": "2606.22082", + "title": "CodeTeam: An LLM-Powered Multi-Agent Framework for Repository-Level Code Generation", + "url": "https://arxiv.org/abs/2606.22082", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Yifei Wang", + "Ruiyin Li", + "Peng Liang", + "Qiong Feng", + "Zengyang Li", + "Mojtaba Shahin", + "Arif Ali Khan" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.22082", + "source": "arxiv", + "source_id": "arxiv:2606.22082", + "pdf_url": "https://arxiv.org/pdf/2606.22082", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.21963", + "title": "Holmes: Multimodal Agentic Diagnosis for Mixed-Language Mobile Crashes at Industrial Scale", + "url": "https://arxiv.org/abs/2606.21963", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Jia Li", + "Wenyuan Ma", + "Ting Peng", + "Haibin Zheng", + "Yuetang Deng" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "rag", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.21963", + "source": "arxiv", + "source_id": "arxiv:2606.21963", + "pdf_url": "https://arxiv.org/pdf/2606.21963", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.21445", + "title": "AutoRAS: Learning Robust Agentic Systems with Primitive Representations", + "url": "https://arxiv.org/abs/2606.21445", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Yang Yue", + "Xuancheng Zhu", + "Yuyang Ma", + "Guoshun Nan", + "Zihan Dou", + "Jingru Shan", + "Congyu Guo", + "Ji Zhang", + "Hua Wang", + "Jingfeng Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.21445", + "source": "arxiv", + "source_id": "arxiv:2606.21445", + "pdf_url": "https://arxiv.org/pdf/2606.21445", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20954", + "title": "Learning What Not to Forget: Long-Horizon Agent Memory from a Few Kilobytes of Learning", + "url": "https://arxiv.org/abs/2606.20954", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Nusrat Jahan Lia", + "Aritra Mazumder" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.20954", + "source": "arxiv", + "source_id": "arxiv:2606.20954", + "pdf_url": "https://arxiv.org/pdf/2606.20954", + "primary_query": "agent-memory" + }, + { + "id": "2606.19980", + "title": "ENPIRE: Agentic Robot Policy Self-Improvement in the Real World", + "url": "https://arxiv.org/abs/2606.19980", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Wenli Xiao", + "Jia Xie", + "Tonghe Zhang", + "Haotian Lin", + "Letian \"Max\" Fu", + "Haoru Xue", + "Jalen Lu", + "Yi Yang", + "Cunxi Dai", + "Zi Wang", + "Jimmy Wu", + "Guanzhi Wang", + "S. Shankar Sastry", + "Ken Goldberg", + "Linxi \"Jim\" Fan", + "Yuke Zhu", + "Guanya Shi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "tool-use" + ], + "arxiv_id": "2606.19980", + "source": "arxiv", + "source_id": "arxiv:2606.19980", + "pdf_url": "https://arxiv.org/pdf/2606.19980", + "primary_query": "coding-agent" + }, + { + "id": "2606.19245", + "title": "TxBench-PP: Analyzing AI Agent Performance on Small-Molecule Preclinical Pharmacology", + "url": "https://arxiv.org/abs/2606.19245", + "published": "2026-06-17", + "updated": "2026-06-18", + "authors": [ + "Hannah Le", + "Ramesh Ramasamy", + "Alex Urrutia", + "Mahsa Yazdani", + "Tim Proctor", + "Kenny Workman" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.19245", + "source": "arxiv", + "source_id": "arxiv:2606.19245", + "pdf_url": "https://arxiv.org/pdf/2606.19245", + "primary_query": "ai-agent" + }, + { + "id": "2606.18619", + "title": "Code-Augur: Agentic Vulnerability Detection via Specification Inference", + "url": "https://arxiv.org/abs/2606.18619", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Zhengxiong Luo", + "Mehtab Zafar", + "Dylan Wolff", + "Abhik Roychoudhury" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-safety", + "computer-use", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.18619", + "source": "arxiv", + "source_id": "arxiv:2606.18619", + "pdf_url": "https://arxiv.org/pdf/2606.18619", + "primary_query": "ai-agent" + }, + { + "id": "2606.19464", + "title": "Deontic Policies for Runtime Governance of Agentic AI Systems", + "url": "https://arxiv.org/abs/2606.19464", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Anupam Joshi", + "Tim Finin", + "Karuna Pande Joshi", + "Lalana Kagal" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.19464", + "source": "arxiv", + "source_id": "arxiv:2606.19464", + "pdf_url": "https://arxiv.org/pdf/2606.19464", + "primary_query": "agentic-ai" + }, + { + "id": "2606.18502", + "title": "Towards Scalable Customization and Deployment of Multi-Agent Systems for Enterprise Applications", + "url": "https://arxiv.org/abs/2606.18502", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Paresh Dashore", + "Shreyas Kulkarni", + "Uttam Gurram", + "Nadia Bathaee", + "Kartik Balasubramaniam", + "Genta Indra Winata", + "Sambit Sahu", + "Shi-Xiong Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "coding-agent", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.18502", + "source": "arxiv", + "source_id": "arxiv:2606.18502", + "pdf_url": "https://arxiv.org/pdf/2606.18502", + "primary_query": "agentic-ai" + }, + { + "id": "2606.18037", + "title": "ProvenanceGuard: Source-Aware Factuality Verification for MCP-Based LLM Agents", + "url": "https://arxiv.org/abs/2606.18037", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Ander Alvarez", + "Santhiya Rajan", + "Samuel Mugel", + "Román Orús" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.18037", + "source": "arxiv", + "source_id": "arxiv:2606.18037", + "pdf_url": "https://arxiv.org/pdf/2606.18037", + "primary_query": "tool-use" + }, + { + "id": "2606.17573", + "title": "Cordon: Semantic Transactions for Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.17573", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Zheng Chen", + "Hanqing Liu", + "Duling Xu", + "Dong Dong", + "Jialin Li", + "Bangzheng Pu", + "Jidong Zhai" + ], + "categories": [ + "cs.OS", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.17573", + "source": "arxiv", + "source_id": "arxiv:2606.17573", + "pdf_url": "https://arxiv.org/pdf/2606.17573", + "primary_query": "tool-use" + }, + { + "id": "2606.18051", + "title": "Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose", + "url": "https://arxiv.org/abs/2606.18051", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Xueping Gao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.18051", + "source": "arxiv", + "source_id": "arxiv:2606.18051", + "pdf_url": "https://arxiv.org/pdf/2606.18051", + "primary_query": "planning-agent" + }, + { + "id": "2606.16295", + "title": "VisualClaw: A Real-Time, Personalized Agent for the Physical World", + "url": "https://arxiv.org/abs/2606.16295", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Haoqin Tu", + "Jianwen Chen", + "Zijun Wang", + "Siwei Han", + "Juncheng Wu", + "Hardy Chen", + "Haonian Ji", + "Kaiwen Xiong", + "Jiaqi Liu", + "Peng Xia", + "Jieru Mei", + "Hongliang Fei", + "Jason Eshraghian", + "Zeyu Zheng", + "Yuyin Zhou", + "Huaxiu Yao", + "Cihang Xie" + ], + "categories": [ + "cs.CV", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "tool-use", + "web-gui-agent" + ], + "arxiv_id": "2606.16295", + "source": "arxiv", + "source_id": "arxiv:2606.16295", + "pdf_url": "https://arxiv.org/pdf/2606.16295", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.16534", + "title": "Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers", + "url": "https://arxiv.org/abs/2606.16534", + "published": "2026-06-15", + "updated": "2026-06-16", + "authors": [ + "Linus Bantel", + "Anna-Lena Roth", + "Jonas Posner", + "Dirk Pflüger" + ], + "categories": [ + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.16534", + "source": "arxiv", + "source_id": "arxiv:2606.16534", + "pdf_url": "https://arxiv.org/pdf/2606.16534", + "primary_query": "tool-use" + }, + { + "id": "2606.15376", + "title": "CoAgent: Concurrency Control for Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.15376", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Hongtao Lyu", + "Dingyan Zhang", + "Mingyu Wu", + "Xingda Wei", + "Haibo Chen" + ], + "categories": [ + "cs.DC", + "cs.AI", + "cs.MA" + ], + "topics": [ + "coding-agent", + "multi-agent", + "planning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.15376", + "source": "arxiv", + "source_id": "arxiv:2606.15376", + "pdf_url": "https://arxiv.org/pdf/2606.15376", + "primary_query": "planning-agent" + }, + { + "id": "2606.14517", + "title": "From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails", + "url": "https://arxiv.org/abs/2606.14517", + "published": "2026-06-12", + "updated": "2026-06-16", + "authors": [ + "Yuguang Zhou", + "Xunguang Wang", + "Pingchuan Ma", + "Zhantong Xue", + "Zhaoyu Wang", + "Shuai Wang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.14517", + "source": "arxiv", + "source_id": "arxiv:2606.14517", + "pdf_url": "https://arxiv.org/pdf/2606.14517", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.14470", + "title": "GitOfThoughts: Version-Controlled Reasoning and Agent Memory You Can Replay, Diff, and Merge", + "url": "https://arxiv.org/abs/2606.14470", + "published": "2026-06-12", + "updated": "2026-06-22", + "authors": [ + "Pavan C Shekar", + "Abhishek H S", + "Aswanth Krishnan" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.14470", + "source": "arxiv", + "source_id": "arxiv:2606.14470", + "pdf_url": "https://arxiv.org/pdf/2606.14470", + "primary_query": "agent-memory" + }, + { + "id": "2606.14106", + "title": "Naive Visual Memory is Not Enough: A Failure-Mode Study of GUI Agents", + "url": "https://arxiv.org/abs/2606.14106", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Seoyoung Choi", + "Minseok Ko", + "Hyunseok Lee", + "Kunwoong Kim", + "Woomin Song", + "Chanseok Jeon", + "Jinwoo Shin" + ], + "categories": [ + "cs.MA", + "cs.CV" + ], + "topics": [ + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.14106", + "source": "arxiv", + "source_id": "arxiv:2606.14106", + "pdf_url": "https://arxiv.org/pdf/2606.14106", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13608", + "title": "AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility", + "url": "https://arxiv.org/abs/2606.13608", + "published": "2026-06-11", + "updated": "2026-06-14", + "authors": [ + "Xiaoyuan Liu", + "Jianhong Tu", + "Yuqi Chen", + "Siyuan Xie", + "Sihan Ren", + "Tianneng Shi", + "Gal Gantar", + "Evan Sandoval", + "Donghyun Lee", + "Daniel Miao", + "Peter J. Gilbert", + "Nick Hynes", + "Mauro Staver", + "Warren He", + "David Marn", + "Andrew Low", + "Xi Zhang", + "Elron Bandel", + "Michal Shmueli-Scheuer", + "Siva Reddy", + "Alexandre Drouin", + "Alexandre Lacoste", + "Ramayya Krishnan", + "Elham Tabassi", + "Yu Su", + "Victor Barres", + "Chenguang Wang", + "Wenbo Guo", + "Dawn Song" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.13608", + "source": "arxiv", + "source_id": "arxiv:2606.13608", + "pdf_url": "https://arxiv.org/pdf/2606.13608", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.13177", + "title": "MemRefine: LLM-Guided Compression for Long-Term Agent Memory", + "url": "https://arxiv.org/abs/2606.13177", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Minjae Kim", + "Jinheon Baek", + "Soyeong Jeong", + "Sung Ju Hwang" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.13177", + "source": "arxiv", + "source_id": "arxiv:2606.13177", + "pdf_url": "https://arxiv.org/pdf/2606.13177", + "primary_query": "agent-memory" + }, + { + "id": "2606.13148", + "title": "TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?", + "url": "https://arxiv.org/abs/2606.13148", + "published": "2026-06-11", + "updated": "2026-07-01", + "authors": [ + "Dat Tien Nguyen", + "Thao Nguyen", + "Fadillah Adamsyah Maani", + "Huy M. Le", + "Muhammad Umer Sheikh", + "Numan Saeed", + "Muhammad Haris Khan", + "Salman Khan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.13148", + "source": "arxiv", + "source_id": "arxiv:2606.13148", + "pdf_url": "https://arxiv.org/pdf/2606.13148", + "primary_query": "tool-use" + }, + { + "id": "2606.28360", + "title": "Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance", + "url": "https://arxiv.org/abs/2606.28360", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Ben Torsion", + "Jun Zhou" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.28360", + "source": "arxiv", + "source_id": "arxiv:2606.28360", + "pdf_url": "https://arxiv.org/pdf/2606.28360", + "primary_query": "rag-agent" + }, + { + "id": "2606.12344", + "title": "Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks", + "url": "https://arxiv.org/abs/2606.12344", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Mengyu Zheng", + "Kai Han", + "Boxun Li", + "Haiyang Xu", + "Yuchuan Tian", + "Wei He", + "Hang Zhou", + "Jianyuan Guo", + "Hailin Hu", + "Lin Ma", + "Chao Xu", + "Guohao Dai", + "Lixue Xia", + "Yunchao Wei", + "Yunhe Wang", + "Yu Wang" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.12344", + "source": "arxiv", + "source_id": "arxiv:2606.12344", + "pdf_url": "https://arxiv.org/pdf/2606.12344", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.11702", + "title": "MedCTA: A Benchmark for Clinical Tool Agents", + "url": "https://arxiv.org/abs/2606.11702", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Tajamul Ashraf", + "Hyewon Jeong", + "Fida Mohammad Thoker", + "Bernard Ghanem" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.11702", + "source": "arxiv", + "source_id": "arxiv:2606.11702", + "pdf_url": "https://arxiv.org/pdf/2606.11702", + "primary_query": "tool-use" + }, + { + "id": "2606.11119", + "title": "TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.11119", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Heming Zou", + "Qi Wang", + "Yun Qu", + "Yuhang Jiang", + "Lizhou Cai", + "Yixiu Mao", + "Ru Peng", + "Xin Xu", + "Weijie Liu", + "Kai Yang", + "Saiyong Yang", + "Xiangyang Ji" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.11119", + "source": "arxiv", + "source_id": "arxiv:2606.11119", + "pdf_url": "https://arxiv.org/pdf/2606.11119", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.10921", + "title": "Trace Only What You Need: Structure-Aware On-Demand Hypergraph Memory for Long-Document Question Answering", + "url": "https://arxiv.org/abs/2606.10921", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Xiangjun Zai", + "Xingyu Tan", + "Chen Chen", + "Xiaoyang Wang", + "Wenjie Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.10921", + "source": "arxiv", + "source_id": "arxiv:2606.10921", + "pdf_url": "https://arxiv.org/pdf/2606.10921", + "primary_query": "rag-agent" + }, + { + "id": "2606.10316", + "title": "TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning", + "url": "https://arxiv.org/abs/2606.10316", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Mingyue Cheng", + "Shuo Yu", + "Daoyu Wang", + "Qingchuan Li", + "Xiaoyu Tao", + "Qingyang Mao", + "Yitong Zhou", + "Qi Liu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.10316", + "source": "arxiv", + "source_id": "arxiv:2606.10316", + "pdf_url": "https://arxiv.org/pdf/2606.10316", + "primary_query": "planning-agent" + }, + { + "id": "2606.09037", + "title": "A Multi-Agent System for IPMSM Design Optimization via an FEA-AI Hybrid Approach", + "url": "https://arxiv.org/abs/2606.09037", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Jinseong Han", + "Sunwoong Yang", + "Namwoo Kang" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.09037", + "source": "arxiv", + "source_id": "arxiv:2606.09037", + "pdf_url": "https://arxiv.org/pdf/2606.09037", + "primary_query": "rag-agent" + }, + { + "id": "2606.09738", + "title": "HDSL: A Hierarchical Domain-Specific Language for Structured 3D Indoor Scene Generation and Localized Editing with LLM Agents", + "url": "https://arxiv.org/abs/2606.09738", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Letian Li", + "Chao Shen", + "Shuzhao Xie", + "Chenghao Gu", + "ZhengXiao He", + "Yu Meng", + "Xin Yang", + "Wenyuan Jiang", + "Zhi Wang" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.09738", + "source": "arxiv", + "source_id": "arxiv:2606.09738", + "pdf_url": "https://arxiv.org/pdf/2606.09738", + "primary_query": "planning-agent" + }, + { + "id": "2606.10209", + "title": "Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.10209", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Abhilasha Lodha", + "Mahsa Pahlavikhah Varnosfaderani", + "Abir Chakraborty", + "Abhinav Mithal" + ], + "categories": [ + "cs.AI", + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.10209", + "source": "arxiv", + "source_id": "arxiv:2606.10209", + "pdf_url": "https://arxiv.org/pdf/2606.10209", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.05711", + "title": "Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems", + "url": "https://arxiv.org/abs/2606.05711", + "published": "2026-06-04", + "updated": "2026-06-05", + "authors": [ + "Yingzhuo Liu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.05711", + "source": "arxiv", + "source_id": "arxiv:2606.05711", + "pdf_url": "https://arxiv.org/pdf/2606.05711", + "primary_query": "language-agent" + }, + { + "id": "2606.06090", + "title": "Beyond Semantic Organization: Memory as Execution State Management for Long-Horizon Agents", + "url": "https://arxiv.org/abs/2606.06090", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Yaoqi Chen", + "Haibin Lai", + "Yuru Feng", + "Chuyu Han", + "Qianxi Zhang", + "Baotong Lu", + "Menghao Li", + "Xinjiang Wang", + "Zhirui Wang", + "Shusen Xu", + "Zengzhong Li", + "Zewen Jin", + "Hao Wu", + "Cheng Li", + "Qi Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2606.06090", + "source": "arxiv", + "source_id": "arxiv:2606.06090", + "pdf_url": "https://arxiv.org/pdf/2606.06090", + "primary_query": "agent-memory" + }, + { + "id": "2606.06388", + "title": "Humans' ALMANAC: A Human Collaboration Dataset of Action-Level Mental Model Annotations for Agent Collaboration", + "url": "https://arxiv.org/abs/2606.06388", + "published": "2026-06-04", + "updated": "2026-06-06", + "authors": [ + "Jiaju Chen", + "Yuxuan Lu", + "Jiayi Su", + "Chaoran Chen", + "Songlin Xiao", + "Zheng Zhang", + "Yun Wang", + "Yunyao Li", + "Jian Zhao", + "Tongshuang Wu", + "Toby Jia-Jun Li", + "Dakuo Wang", + "Bingsheng Yao" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.06388", + "source": "arxiv", + "source_id": "arxiv:2606.06388", + "pdf_url": "https://arxiv.org/pdf/2606.06388", + "primary_query": "planning-agent" + }, + { + "id": "2606.05241", + "title": "Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation", + "url": "https://arxiv.org/abs/2606.05241", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Yongjie Wang", + "Xinyue Zhang", + "Kunhong Yao", + "Zhiwei Zeng", + "Kaisong Song", + "Jun Lin", + "Zhiqi Shen" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.05241", + "source": "arxiv", + "source_id": "arxiv:2606.05241", + "pdf_url": "https://arxiv.org/pdf/2606.05241", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.03197", + "title": "MemTrain: Self-Supervised Context Memory Training", + "url": "https://arxiv.org/abs/2606.03197", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Ziheng Li", + "Xingrun Xing", + "Haoqing Wang", + "Zhi-Hong Deng", + "Yehui Tang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.03197", + "source": "arxiv", + "source_id": "arxiv:2606.03197", + "pdf_url": "https://arxiv.org/pdf/2606.03197", + "primary_query": "agent-memory" + }, + { + "id": "2606.02497", + "title": "Bridging the Last Mile of Time Series Forecasting with LLM Agents", + "url": "https://arxiv.org/abs/2606.02497", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Yuhua Liao", + "Zetian Wang", + "Qiangqiang Nie", + "Zhenhua Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.02497", + "source": "arxiv", + "source_id": "arxiv:2606.02497", + "pdf_url": "https://arxiv.org/pdf/2606.02497", + "primary_query": "planning-agent" + }, + { + "id": "2606.01222", + "title": "RAG-driven Multi-Agent LLM Framework with Task Decomposition for Beyond 5G Auto-Configuration", + "url": "https://arxiv.org/abs/2606.01222", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "İrşat Emin Sarıdaş", + "Onur Salan", + "Ali Görçin", + "Ibrahim Hokelek", + "Hakan Ali Çırpan" + ], + "categories": [ + "eess.SP" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.01222", + "source": "arxiv", + "source_id": "arxiv:2606.01222", + "pdf_url": "https://arxiv.org/pdf/2606.01222", + "primary_query": "rag-agent" + }, + { + "id": "2606.00619", + "title": "MemPro: Agentic Memory Systems as Evolvable Programs", + "url": "https://arxiv.org/abs/2606.00619", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Qingshan Liu", + "Guoqing Wang", + "Wen Wu", + "Jingqi Huang", + "Xinqi Tao", + "Dejia Song", + "Jie Zhou", + "Liang He" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.00619", + "source": "arxiv", + "source_id": "arxiv:2606.00619", + "pdf_url": "https://arxiv.org/pdf/2606.00619", + "primary_query": "agent-memory" + }, + { + "id": "2606.00922", + "title": "A Machine-to-Machine Knowledge-Guided LLM Agent for Generalizable Radiotherapy Treatment Planning", + "url": "https://arxiv.org/abs/2606.00922", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Md Mainul Abrar", + "Xun Jia", + "Yujie Chi" + ], + "categories": [ + "physics.med-ph", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.00922", + "source": "arxiv", + "source_id": "arxiv:2606.00922", + "pdf_url": "https://arxiv.org/pdf/2606.00922", + "primary_query": "planning-agent" + }, + { + "id": "2606.07595", + "title": "VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents", + "url": "https://arxiv.org/abs/2606.07595", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Youting Wang", + "Yuan Tang", + "Yitian Qian", + "Chen Zhao" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.07595", + "source": "arxiv", + "source_id": "arxiv:2606.07595", + "pdf_url": "https://arxiv.org/pdf/2606.07595", + "primary_query": "language-agent" + }, + { + "id": "2606.00341", + "title": "ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use", + "url": "https://arxiv.org/abs/2606.00341", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Jeremy Tien", + "Abishek Anand", + "Yu-Rou Tuan", + "Yuchen Shen", + "J. Zico Kolter", + "Aran Nayebi" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.00341", + "source": "arxiv", + "source_id": "arxiv:2606.00341", + "pdf_url": "https://arxiv.org/pdf/2606.00341", + "primary_query": "agent-safety" + }, + { + "id": "2605.31377", + "title": "DynaTree: Dynamic Agentic Retrieval Tree for Time-Sensitive News Retrieval", + "url": "https://arxiv.org/abs/2605.31377", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Siyuan Qi", + "Xinyuan Wang", + "Yingxuan Yang", + "Haochuan Guo", + "Jianghao Lin", + "Weiwen Liu", + "Yong Yu", + "Weinan Zhang" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.31377", + "source": "arxiv", + "source_id": "arxiv:2605.31377", + "pdf_url": "https://arxiv.org/pdf/2605.31377", + "primary_query": "rag-agent" + }, + { + "id": "2605.30947", + "title": "Extending AI for Research to the Humanities: A Multi-Agent Framework for Evidence-Grounded Scholarship", + "url": "https://arxiv.org/abs/2605.30947", + "published": "2026-05-29", + "updated": "2026-06-03", + "authors": [ + "Yating Pan", + "Jiajun Zhang", + "Jun Wang", + "Qi Su" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.30947", + "source": "arxiv", + "source_id": "arxiv:2605.30947", + "pdf_url": "https://arxiv.org/pdf/2605.30947", + "primary_query": "rag-agent" + }, + { + "id": "2605.29960", + "title": "Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction", + "url": "https://arxiv.org/abs/2605.29960", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Hongtao Wang", + "Se Yang", + "Yu Chen", + "Puzhuo Liu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.29960", + "source": "arxiv", + "source_id": "arxiv:2605.29960", + "pdf_url": "https://arxiv.org/pdf/2605.29960", + "primary_query": "agent-memory" + }, + { + "id": "2605.25435", + "title": "Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures", + "url": "https://arxiv.org/abs/2605.25435", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Yuntao Wang", + "Jianle Ba", + "Han Liu", + "Yanghe Pan", + "Jintao Wei", + "Zhou Su", + "Tom H. Luan", + "Linkang Du" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.25435", + "source": "arxiv", + "source_id": "arxiv:2605.25435", + "pdf_url": "https://arxiv.org/pdf/2605.25435", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.24812", + "title": "CoRe-Code: Collaborative Reinforcement Learning for Code Generation", + "url": "https://arxiv.org/abs/2605.24812", + "published": "2026-05-24", + "updated": "2026-05-24", + "authors": [ + "Zhihao Dou", + "Qinjian Zhao", + "Zhongwei Wan", + "Xiaoyu Xia", + "Sumon Biswas" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.24812", + "source": "arxiv", + "source_id": "arxiv:2605.24812", + "pdf_url": "https://arxiv.org/pdf/2605.24812", + "primary_query": "planning-agent" + }, + { + "id": "2605.23067", + "title": "What Training Data Teaches RL Memory Agents: An Empirical Study of Curriculum Effects in Memory-Augmented QA", + "url": "https://arxiv.org/abs/2605.23067", + "published": "2026-05-21", + "updated": "2026-05-21", + "authors": [ + "Xinjie He", + "Zhiyuan Lin", + "Su Liu", + "Jialun Wu", + "Qiyang Xie", + "Weikai Zhou", + "Shuai Xiao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.23067", + "source": "arxiv", + "source_id": "arxiv:2605.23067", + "pdf_url": "https://arxiv.org/pdf/2605.23067", + "primary_query": "agent-memory" + }, + { + "id": "2605.20616", + "title": "Auto-Dreamer: Learning Offline Memory Consolidation for Language Agents", + "url": "https://arxiv.org/abs/2605.20616", + "published": "2026-05-20", + "updated": "2026-05-20", + "authors": [ + "Chongrui Ye", + "Yuxiang Liu", + "Yu Wang", + "Haofei Yu", + "Yining Zhao", + "Ge Liu", + "Julian McAuley", + "Jiaxuan You" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "language-agent" + ], + "arxiv_id": "2605.20616", + "source": "arxiv", + "source_id": "arxiv:2605.20616", + "pdf_url": "https://arxiv.org/pdf/2605.20616", + "primary_query": "agent-memory" + }, + { + "id": "2605.16233", + "title": "FORGE: Self-Evolving Agent Memory With No Weight Updates via Population Broadcast", + "url": "https://arxiv.org/abs/2605.16233", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Igor Bogdanov", + "Chung-Horng Lung", + "Thomas Kunz", + "Jie Gao", + "Adrian Taylor", + "Marzia Zaman" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA", + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.16233", + "source": "arxiv", + "source_id": "arxiv:2605.16233", + "pdf_url": "https://arxiv.org/pdf/2605.16233", + "primary_query": "agent-memory" + }, + { + "id": "2605.15759", + "title": "DimMem: Dimensional Structuring for Efficient Long-Term Agent Memory", + "url": "https://arxiv.org/abs/2605.15759", + "published": "2026-05-15", + "updated": "2026-05-24", + "authors": [ + "Wentao Qiu", + "Haotian Hu", + "Fanyi Wang", + "Jinwei Kong", + "Yu Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.15759", + "source": "arxiv", + "source_id": "arxiv:2605.15759", + "pdf_url": "https://arxiv.org/pdf/2605.15759", + "primary_query": "agent-memory" + }, + { + "id": "2605.15710", + "title": "SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory", + "url": "https://arxiv.org/abs/2605.15710", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Huacan Chai", + "Yukai Wang", + "Yingxuan Yang", + "Dan Peng", + "Yuanyi Song", + "Zhihui Fu", + "Weiwen Liu", + "Jianghao Lin", + "Jun Wang", + "Weinan Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.15710", + "source": "arxiv", + "source_id": "arxiv:2605.15710", + "pdf_url": "https://arxiv.org/pdf/2605.15710", + "primary_query": "agent-memory" + }, + { + "id": "2605.15128", + "title": "MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory", + "url": "https://arxiv.org/abs/2605.15128", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Minghao Guo", + "Qingyue Jiao", + "Zeru Shi", + "Yihao Quan", + "Boxuan Zhang", + "Danrui Li", + "Liwei Che", + "Wujiang Xu", + "Shilong Liu", + "Zirui Liu", + "Mubbasir Kapadia", + "Vladimir Pavlovic", + "Jiang Liu", + "Mengdi Wang", + "Yiyu Shi", + "Dimitris N. Metaxas", + "Ruixiang Tang" + ], + "categories": [ + "cs.CV", + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.15128", + "source": "arxiv", + "source_id": "arxiv:2605.15128", + "pdf_url": "https://arxiv.org/pdf/2605.15128", + "primary_query": "agent-memory" + }, + { + "id": "2605.14906", + "title": "MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models", + "url": "https://arxiv.org/abs/2605.14906", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Xiyu Ren", + "Zhaowei Wang", + "Yiming Du", + "Zhongwei Xie", + "Chi Liu", + "Xinlin Yang", + "Haoyue Feng", + "Wenjun Pan", + "Tianshi Zheng", + "Baixuan Xu", + "Zhengnan Li", + "Yangqiu Song", + "Ginny Wong", + "Simon See" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.14906", + "source": "arxiv", + "source_id": "arxiv:2605.14906", + "pdf_url": "https://arxiv.org/pdf/2605.14906", + "primary_query": "agent-memory" + }, + { + "id": "2605.14932", + "title": "Toward Securing AI Agents Like Operating Systems", + "url": "https://arxiv.org/abs/2605.14932", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Lukas Pirch", + "Micha Horlboge", + "Patrick Großmann", + "Syeda Mahnur Asif", + "Klim Kireev", + "Thorsten Holz", + "Konrad Rieck" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14932", + "source": "arxiv", + "source_id": "arxiv:2605.14932", + "pdf_url": "https://arxiv.org/pdf/2605.14932", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.11946", + "title": "Counterfactual Trace Auditing of LLM Agent Skills", + "url": "https://arxiv.org/abs/2605.11946", + "published": "2026-05-12", + "updated": "2026-05-28", + "authors": [ + "Xiaolin Zhou", + "Jinbo Liu", + "Li Li", + "Ryan A. Rossi", + "Xiyang Hu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.11946", + "source": "arxiv", + "source_id": "arxiv:2605.11946", + "pdf_url": "https://arxiv.org/pdf/2605.11946", + "primary_query": "planning-agent" + }, + { + "id": "2605.08964", + "title": "Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents", + "url": "https://arxiv.org/abs/2605.08964", + "published": "2026-05-09", + "updated": "2026-05-09", + "authors": [ + "Carol Xuan Long" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.08964", + "source": "arxiv", + "source_id": "arxiv:2605.08964", + "pdf_url": "https://arxiv.org/pdf/2605.08964", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.07830", + "title": "CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios", + "url": "https://arxiv.org/abs/2605.07830", + "published": "2026-05-08", + "updated": "2026-05-08", + "authors": [ + "Taein Lim", + "Seongyong Ju", + "Munhyeok Kim", + "Hyunjun Kim", + "Hoki Kim" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.07830", + "source": "arxiv", + "source_id": "arxiv:2605.07830", + "pdf_url": "https://arxiv.org/pdf/2605.07830", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.04808", + "title": "DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents", + "url": "https://arxiv.org/abs/2605.04808", + "published": "2026-05-06", + "updated": "2026-05-06", + "authors": [ + "Zhaorun Chen", + "Xun Liu", + "Haibo Tong", + "Chengquan Guo", + "Yuzhou Nie", + "Jiawei Zhang", + "Mintong Kang", + "Chejian Xu", + "Qichang Liu", + "Xiaogeng Liu", + "Tianneng Shi", + "Chaowei Xiao", + "Sanmi Koyejo", + "Percy Liang", + "Wenbo Guo", + "Dawn Song", + "Bo Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.04808", + "source": "arxiv", + "source_id": "arxiv:2605.04808", + "pdf_url": "https://arxiv.org/pdf/2605.04808", + "primary_query": "agent-safety" + }, + { + "id": "2605.01644", + "title": "Toward a Principled Framework for Agent Safety Measurement", + "url": "https://arxiv.org/abs/2605.01644", + "published": "2026-05-02", + "updated": "2026-05-02", + "authors": [ + "Shuyi Lin", + "Anshuman Suri", + "Alina Oprea", + "Cheng Tan" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.01644", + "source": "arxiv", + "source_id": "arxiv:2605.01644", + "pdf_url": "https://arxiv.org/pdf/2605.01644", + "primary_query": "agent-safety" + }, + { + "id": "2605.00741", + "title": "Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems", + "url": "https://arxiv.org/abs/2605.00741", + "published": "2026-05-01", + "updated": "2026-05-01", + "authors": [ + "Saeid Jamshidi", + "Foutse Khomh", + "Carol Fung", + "Kawser Wazed Nafi" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.00741", + "source": "arxiv", + "source_id": "arxiv:2605.00741", + "pdf_url": "https://arxiv.org/pdf/2605.00741", + "primary_query": "agent-safety" + }, + { + "id": "2604.24212", + "title": "Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis", + "url": "https://arxiv.org/abs/2604.24212", + "published": "2026-04-27", + "updated": "2026-04-27", + "authors": [ + "Jiahong Xiang", + "Xiaoyang Xu", + "Xiaopan Chu", + "Hongliang Tian", + "Yuqun Zhang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.24212", + "source": "arxiv", + "source_id": "arxiv:2604.24212", + "pdf_url": "https://arxiv.org/pdf/2604.24212", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.23210", + "title": "Discovering Agentic Safety Specifications from 1-Bit Danger Signals", + "url": "https://arxiv.org/abs/2604.23210", + "published": "2026-04-25", + "updated": "2026-04-25", + "authors": [ + "Víctor Gallego" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.23210", + "source": "arxiv", + "source_id": "arxiv:2604.23210", + "pdf_url": "https://arxiv.org/pdf/2604.23210", + "primary_query": "agent-safety" + }, + { + "id": "2604.13954", + "title": "HINTBench: Horizon-agent Intrinsic Non-attack Trajectory Benchmark", + "url": "https://arxiv.org/abs/2604.13954", + "published": "2026-04-15", + "updated": "2026-04-15", + "authors": [ + "Jiacheng Wang", + "Jinchang Hou", + "Fabian Wang", + "Ping Jian", + "Chenfu Bao", + "Zhonghou Lv" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.13954", + "source": "arxiv", + "source_id": "arxiv:2604.13954", + "pdf_url": "https://arxiv.org/pdf/2604.13954", + "primary_query": "agent-safety" + }, + { + "id": "2604.13536", + "title": "Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy", + "url": "https://arxiv.org/abs/2604.13536", + "published": "2026-04-15", + "updated": "2026-04-16", + "authors": [ + "Shawn Wanxiang Zhong", + "Junxuan Liao", + "Jing Liu", + "Mai Zheng", + "Andrea C. Arpaci-Dusseau", + "Remzi H. Arpaci-Dusseau" + ], + "categories": [ + "cs.OS" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.13536", + "source": "arxiv", + "source_id": "arxiv:2604.13536", + "pdf_url": "https://arxiv.org/pdf/2604.13536", + "primary_query": "agent-safety" + }, + { + "id": "2604.13298", + "title": "Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection", + "url": "https://arxiv.org/abs/2604.13298", + "published": "2026-04-14", + "updated": "2026-04-14", + "authors": [ + "Sujan Ghimire", + "Parsa Mirfasihi", + "Muhtasim Alam Chowdhury", + "Veeramani Pugazhenthi", + "Harish Kumar Dharavath", + "Farshad Firouzi", + "Rozhin Yasaei", + "Pratik Satam", + "Soheil Salehi" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.13298", + "source": "arxiv", + "source_id": "arxiv:2604.13298", + "pdf_url": "https://arxiv.org/pdf/2604.13298", + "primary_query": "agent-safety" + }, + { + "id": "2605.28835", + "title": "GenesisFunc: Multi-Agent Data Generation for Accurate and Generalizable Function-Calling", + "url": "https://arxiv.org/abs/2605.28835", + "published": "2026-04-10", + "updated": "2026-04-10", + "authors": [ + "Hao-Xiang Xu", + "Chong Deng", + "Jiaqing Liu", + "Wen Wang", + "Qian Chen", + "Lujia Bao", + "Xiangang Li", + "Zhen-Hua Ling" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.28835", + "source": "arxiv", + "source_id": "arxiv:2605.28835", + "pdf_url": "https://arxiv.org/pdf/2605.28835", + "primary_query": "function-calling" + }, + { + "id": "2605.00845", + "title": "Graph Query Generation with Constraint-guided Large Language Agents", + "url": "https://arxiv.org/abs/2605.00845", + "published": "2026-04-09", + "updated": "2026-04-09", + "authors": [ + "Mengying Wang", + "Nicolaas Jedema", + "Rahul Pandey", + "RaviKiran Krishnan", + "Jens Lehmann", + "Yinghui Wu" + ], + "categories": [ + "cs.DB", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.00845", + "source": "arxiv", + "source_id": "arxiv:2605.00845", + "pdf_url": "https://arxiv.org/pdf/2605.00845", + "primary_query": "language-agent" + }, + { + "id": "2604.26959", + "title": "CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs", + "url": "https://arxiv.org/abs/2604.26959", + "published": "2026-04-07", + "updated": "2026-04-07", + "authors": [ + "Elham Nasarian", + "Abhilash Neog", + "Kwok-Leung Tsui", + "Niyousha HosseiniChimeh" + ], + "categories": [ + "cs.CY", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.26959", + "source": "arxiv", + "source_id": "arxiv:2604.26959", + "pdf_url": "https://arxiv.org/pdf/2604.26959", + "primary_query": "agent-safety" + }, + { + "id": "2604.04131", + "title": "Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents", + "url": "https://arxiv.org/abs/2604.04131", + "published": "2026-04-05", + "updated": "2026-04-05", + "authors": [ + "Paulo Akira F. Enabe" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.04131", + "source": "arxiv", + "source_id": "arxiv:2604.04131", + "pdf_url": "https://arxiv.org/pdf/2604.04131", + "primary_query": "language-agent" + }, + { + "id": "2603.28428", + "title": "Synergy: A Next-Generation General-Purpose Agent for Open Agentic Web", + "url": "https://arxiv.org/abs/2603.28428", + "published": "2026-03-30", + "updated": "2026-03-30", + "authors": [ + "Xiaohang Nie", + "Zihan Guo", + "Kezhuo Yang", + "Zhichong Zheng", + "Bochen Ge", + "Shuai Pan", + "Zeyi Chen", + "Youling Xiang", + "Yu Zhang", + "Weiwen Liu", + "Yuanjian Zhou", + "Weinan Zhang" + ], + "categories": [ + "cs.CY", + "cs.MA" + ], + "topics": [ + "coding-agent", + "embodied-agent", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.28428", + "source": "arxiv", + "source_id": "arxiv:2603.28428", + "pdf_url": "https://arxiv.org/pdf/2603.28428", + "primary_query": "function-calling" + }, + { + "id": "2603.28166", + "title": "Evaluating Privilege Usage of Agents with Real-World Tools", + "url": "https://arxiv.org/abs/2603.28166", + "published": "2026-03-30", + "updated": "2026-04-20", + "authors": [ + "Quan Zhang", + "Lianhang Fu", + "Lvsi Lian", + "Gwihwan Go", + "Yujue Wang", + "Chijin Zhou", + "Yu Jiang", + "Geguang Pu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.28166", + "source": "arxiv", + "source_id": "arxiv:2603.28166", + "pdf_url": "https://arxiv.org/pdf/2603.28166", + "primary_query": "agent-safety" + }, + { + "id": "2603.21564", + "title": "Toward a Theory of Hierarchical Memory for Language Agents", + "url": "https://arxiv.org/abs/2603.21564", + "published": "2026-03-23", + "updated": "2026-03-23", + "authors": [ + "Yashar Talebirad", + "Ali Parsaee", + "Csongor Y. Szepesvari", + "Amirhossein Nadiri", + "Osmar Zaiane" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.IT", + "cs.SI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.21564", + "source": "arxiv", + "source_id": "arxiv:2603.21564", + "pdf_url": "https://arxiv.org/pdf/2603.21564", + "primary_query": "language-agent" + }, + { + "id": "2603.21357", + "title": "AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling", + "url": "https://arxiv.org/abs/2603.21357", + "published": "2026-03-22", + "updated": "2026-05-10", + "authors": [ + "Liang Ding" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "computer-use", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.21357", + "source": "arxiv", + "source_id": "arxiv:2603.21357", + "pdf_url": "https://arxiv.org/pdf/2603.21357", + "primary_query": "language-agent" + }, + { + "id": "2603.05553", + "title": "EigenData: A Self-Evolving Multi-Agent Platform for Function-Calling Data Synthesis, Auditing, and Repair", + "url": "https://arxiv.org/abs/2603.05553", + "published": "2026-03-05", + "updated": "2026-03-05", + "authors": [ + "Jiaao Chen", + "Jingyuan Qi", + "Mingye Gao", + "Wei-Chen Wang", + "Hanrui Wang", + "Di Jin" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.05553", + "source": "arxiv", + "source_id": "arxiv:2603.05553", + "pdf_url": "https://arxiv.org/pdf/2603.05553", + "primary_query": "function-calling" + }, + { + "id": "2602.07652", + "title": "Agent-Fence: Mapping Security Vulnerabilities Across Deep Research Agents", + "url": "https://arxiv.org/abs/2602.07652", + "published": "2026-02-07", + "updated": "2026-02-07", + "authors": [ + "Sai Puppala", + "Ismail Hossain", + "Md Jahangir Alam", + "Yoonpyo Lee", + "Jay Yoo", + "Tanzim Ahad", + "Syed Bahauddin Alam", + "Sajedul Talukder" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.07652", + "source": "arxiv", + "source_id": "arxiv:2602.07652", + "pdf_url": "https://arxiv.org/pdf/2602.07652", + "primary_query": "agent-safety" + }, + { + "id": "2602.05115", + "title": "SocialVeil: Probing Social Intelligence of Language Agents under Communication Barriers", + "url": "https://arxiv.org/abs/2602.05115", + "published": "2026-02-04", + "updated": "2026-02-04", + "authors": [ + "Keyang Xuan", + "Pengda Wang", + "Chongrui Ye", + "Haofei Yu", + "Tal August", + "Jiaxuan You" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.05115", + "source": "arxiv", + "source_id": "arxiv:2602.05115", + "pdf_url": "https://arxiv.org/pdf/2602.05115", + "primary_query": "language-agent" + }, + { + "id": "2602.03786", + "title": "AOrchestra: Automating Sub-Agent Creation for Agentic Orchestration", + "url": "https://arxiv.org/abs/2602.03786", + "published": "2026-02-03", + "updated": "2026-02-07", + "authors": [ + "Jianhao Ruan", + "Zhihao Xu", + "Yiran Peng", + "Fashen Ren", + "Zhaoyang Yu", + "Xinbing Liang", + "Jinyu Xiang", + "Yongru Chen", + "Bang Liu", + "Chenglin Wu", + "Yuyu Luo", + "Jiayi Zhang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.03786", + "source": "arxiv", + "source_id": "arxiv:2602.03786", + "pdf_url": "https://arxiv.org/pdf/2602.03786", + "primary_query": "language-agent" + }, + { + "id": "2601.00268", + "title": "Beyond Perfect APIs: A Comprehensive Evaluation of LLM Agents Under Real-World API Complexity", + "url": "https://arxiv.org/abs/2601.00268", + "published": "2026-01-01", + "updated": "2026-01-01", + "authors": [ + "Doyoung Kim", + "Zhiwei Ren", + "Jie Hao", + "Zhongkai Sun", + "Lichao Wang", + "Xiyao Ma", + "Zack Ye", + "Xu Han", + "Jun Yin", + "Heng Ji", + "Wei Shen", + "Xing Fan", + "Benjamin Yao", + "Chenlei Guo" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.00268", + "source": "arxiv", + "source_id": "arxiv:2601.00268", + "pdf_url": "https://arxiv.org/pdf/2601.00268", + "primary_query": "function-calling" + }, + { + "id": "2511.22138", + "title": "TinyLLM: Evaluation and Optimization of Small Language Models for Agentic Tasks on Edge Devices", + "url": "https://arxiv.org/abs/2511.22138", + "published": "2025-11-27", + "updated": "2025-11-27", + "authors": [ + "Mohd Ariful Haque", + "Fahad Rahman", + "Kishor Datta Gupta", + "Khalil Shujaee", + "Roy George" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2511.22138", + "source": "arxiv", + "source_id": "arxiv:2511.22138", + "pdf_url": "https://arxiv.org/pdf/2511.22138", + "primary_query": "function-calling" + }, + { + "id": "2511.11169", + "title": "Refine and Align: Confidence Calibration through Multi-Agent Interaction in VQA", + "url": "https://arxiv.org/abs/2511.11169", + "published": "2025-11-14", + "updated": "2025-11-14", + "authors": [ + "Ayush Pandey", + "Jai Bardhan", + "Ishita Jain", + "Ramya S Hebbalaguppe", + "Rohan Raju Dhanakshirur", + "Lovekesh Vig" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2511.11169", + "source": "arxiv", + "source_id": "arxiv:2511.11169", + "pdf_url": "https://arxiv.org/pdf/2511.11169", + "primary_query": "function-calling" + }, + { + "id": "2510.21524", + "title": "EU-Agent-Bench: Measuring Illegal Behavior of LLM Agents Under EU Law", + "url": "https://arxiv.org/abs/2510.21524", + "published": "2025-10-24", + "updated": "2025-10-24", + "authors": [ + "Ilija Lichkovski", + "Alexander Müller", + "Mariam Ibrahim", + "Tiwai Mhundwa" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.21524", + "source": "arxiv", + "source_id": "arxiv:2510.21524", + "pdf_url": "https://arxiv.org/pdf/2510.21524", + "primary_query": "function-calling" + }, + { + "id": "2509.08863", + "title": "GeoJSON Agents:A Multi-Agent LLM Architecture for Geospatial Analysis-Function Calling vs Code Generation", + "url": "https://arxiv.org/abs/2509.08863", + "published": "2025-09-10", + "updated": "2025-12-03", + "authors": [ + "Qianqian Luo", + "Qingming Lin", + "Liuchang Xu", + "Sensen Wu", + "Ruichen Mao", + "Chao Wang", + "Hailin Feng", + "Bo Huang", + "Zhenhong Du" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.08863", + "source": "arxiv", + "source_id": "arxiv:2509.08863", + "pdf_url": "https://arxiv.org/pdf/2509.08863", + "primary_query": "function-calling" + }, + { + "id": "2508.11027", + "title": "Hell or High Water: Evaluating Agentic Recovery from External Failures", + "url": "https://arxiv.org/abs/2508.11027", + "published": "2025-08-14", + "updated": "2025-08-14", + "authors": [ + "Andrew Wang", + "Sophia Hager", + "Adi Asija", + "Daniel Khashabi", + "Nicholas Andrews" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 16, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.11027", + "source": "arxiv", + "source_id": "arxiv:2508.11027", + "pdf_url": "https://arxiv.org/pdf/2508.11027", + "primary_query": "function-calling" + }, + { + "id": "2607.06223", + "title": "Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents", + "url": "https://arxiv.org/abs/2607.06223", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Yijun Zhang", + "Fan Xu", + "Jiaxin Ding", + "Yule Xie", + "Shiqing Gao", + "Xin Ding", + "Haoxiang Zhang", + "Luoyi Fu", + "Xinbing Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.06223", + "source": "arxiv", + "source_id": "arxiv:2607.06223", + "pdf_url": "https://arxiv.org/pdf/2607.06223", + "primary_query": "llm-agent" + }, + { + "id": "2607.05915", + "title": "PCBWorld: A Benchmark Environment for Engine-Grounded PCB Design Automation", + "url": "https://arxiv.org/abs/2607.05915", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Hyungseok Song", + "Junseok Park", + "Won-Seok Choi", + "Seohui Bae", + "Han-Seul Jeong", + "Youngjoon Park", + "Soonyoung Lee" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "llm-agent", + "tool-use" + ], + "arxiv_id": "2607.05915", + "source": "arxiv", + "source_id": "arxiv:2607.05915", + "pdf_url": "https://arxiv.org/pdf/2607.05915", + "primary_query": "agentic-ai" + }, + { + "id": "2607.05805", + "title": "Onnes: A Physics-Grounded Multi-Agent LLM Simulator for Cryogenic Fault Diagnosis in Quantum Computing Infrastructure", + "url": "https://arxiv.org/abs/2607.05805", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Praneeth Narisetty", + "Uday Kumar Reddy Kattamanchi", + "Shiva Nagendra Babu Kore" + ], + "categories": [ + "cs.AI", + "cs.LG", + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.05805", + "source": "arxiv", + "source_id": "arxiv:2607.05805", + "pdf_url": "https://arxiv.org/pdf/2607.05805", + "primary_query": "llm-agent" + }, + { + "id": "2607.05743", + "title": "The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities", + "url": "https://arxiv.org/abs/2607.05743", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Mohammadreza Rashidi" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2607.05743", + "source": "arxiv", + "source_id": "arxiv:2607.05743", + "pdf_url": "https://arxiv.org/pdf/2607.05743", + "primary_query": "agentic-ai" + }, + { + "id": "2607.05690", + "title": "Memory in the Loop: In-Process Retrieval as ExtendedWorking Memory for Language Agents", + "url": "https://arxiv.org/abs/2607.05690", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Yusuf Khan", + "Carlo Lipizzi" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2607.05690", + "source": "arxiv", + "source_id": "arxiv:2607.05690", + "pdf_url": "https://arxiv.org/pdf/2607.05690", + "primary_query": "language-agent" + }, + { + "id": "2607.05055", + "title": "Toward Trustworthy Large Language Model Agents in Healthcare", + "url": "https://arxiv.org/abs/2607.05055", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Hadi Hasan", + "Safaa Salman", + "Adam Tai Abou Dargham", + "Ammar Mohanna", + "Ali Chehab" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling", + "rag-agent" + ], + "arxiv_id": "2607.05055", + "source": "arxiv", + "source_id": "arxiv:2607.05055", + "pdf_url": "https://arxiv.org/pdf/2607.05055", + "primary_query": "function-calling" + }, + { + "id": "2607.04089", + "title": "PLACEMEM: Toward a Compute-Aware Memory Plane for Lifelong Agents", + "url": "https://arxiv.org/abs/2607.04089", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Sukanta Ganguly" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2607.04089", + "source": "arxiv", + "source_id": "arxiv:2607.04089", + "pdf_url": "https://arxiv.org/pdf/2607.04089", + "primary_query": "agent-memory" + }, + { + "id": "2607.03702", + "title": "Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry", + "url": "https://arxiv.org/abs/2607.03702", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Weiyang Guo", + "Zesheng Shi", + "Longhui Zhang", + "Zeen Zhu", + "Min Zhang", + "Jing Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.03702", + "source": "arxiv", + "source_id": "arxiv:2607.03702", + "pdf_url": "https://arxiv.org/pdf/2607.03702", + "primary_query": "llm-agent" + }, + { + "id": "2607.03695", + "title": "Social Networks of LLM Agents", + "url": "https://arxiv.org/abs/2607.03695", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Kaixuan Liu", + "Guojun Xiong", + "Weinan Zhang", + "Shengpu Tang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.03695", + "source": "arxiv", + "source_id": "arxiv:2607.03695", + "pdf_url": "https://arxiv.org/pdf/2607.03695", + "primary_query": "llm-agent" + }, + { + "id": "2607.03821", + "title": "DualView: Preventing Indirect Prompt Injection in Personal AI Agents", + "url": "https://arxiv.org/abs/2607.03821", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Juhee Kim", + "Woohyuk Choi", + "Taehyun Kang", + "Youngmin Kim", + "Byoungyoung Lee" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.03821", + "source": "arxiv", + "source_id": "arxiv:2607.03821", + "pdf_url": "https://arxiv.org/pdf/2607.03821", + "primary_query": "ai-agent" + }, + { + "id": "2607.04009", + "title": "PhysMiner: An Agentic AI Framework for Discovering Turbulence Physics", + "url": "https://arxiv.org/abs/2607.04009", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Jiawei Chen", + "Han Gao", + "Ping He" + ], + "categories": [ + "physics.flu-dyn" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.04009", + "source": "arxiv", + "source_id": "arxiv:2607.04009", + "pdf_url": "https://arxiv.org/pdf/2607.04009", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03968", + "title": "Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents", + "url": "https://arxiv.org/abs/2607.03968", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Abhishek Kumar", + "Carsten Maple" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03968", + "source": "arxiv", + "source_id": "arxiv:2607.03968", + "pdf_url": "https://arxiv.org/pdf/2607.03968", + "primary_query": "coding-agent" + }, + { + "id": "2607.02846", + "title": "Object-Centric Environment Modeling for Agentic Tasks", + "url": "https://arxiv.org/abs/2607.02846", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Yiyang Li", + "Tianyi Ma", + "Zehong Wang", + "Yijun Ma", + "Yanfang Ye" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.02846", + "source": "arxiv", + "source_id": "arxiv:2607.02846", + "pdf_url": "https://arxiv.org/pdf/2607.02846", + "primary_query": "llm-agent" + }, + { + "id": "2607.03269", + "title": "Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks", + "url": "https://arxiv.org/abs/2607.03269", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Haoxiang Luo", + "Yinqiu Liu", + "Ruichen Zhang", + "Guangyuan Liu", + "Gang Sun", + "Hongfang Yu", + "Zhu Han", + "Dong In Kim" + ], + "categories": [ + "cs.NI" + ], + "topics": [ + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.03269", + "source": "arxiv", + "source_id": "arxiv:2607.03269", + "pdf_url": "https://arxiv.org/pdf/2607.03269", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02911", + "title": "CoACT: Action-Preserving Observation Compression for Coding Agents", + "url": "https://arxiv.org/abs/2607.02911", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Haorui Chen", + "Yuancheng Zhu", + "Yitong Zhang", + "Jia Li" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.02911", + "source": "arxiv", + "source_id": "arxiv:2607.02911", + "pdf_url": "https://arxiv.org/pdf/2607.02911", + "primary_query": "coding-agent" + }, + { + "id": "2607.01767", + "title": "Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts", + "url": "https://arxiv.org/abs/2607.01767", + "published": "2026-07-02", + "updated": "2026-07-05", + "authors": [ + "Xinyuan Song", + "Zekun Cai" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2607.01767", + "source": "arxiv", + "source_id": "arxiv:2607.01767", + "pdf_url": "https://arxiv.org/pdf/2607.01767", + "primary_query": "language-agent" + }, + { + "id": "2607.01709", + "title": "COMFYCLAW: Self-Evolving Skill Harnesses for Image Generation Workflows", + "url": "https://arxiv.org/abs/2607.01709", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Zongxia Li", + "Dawei Liu", + "Fuxiao Liu", + "Yuhang Zhou", + "Xiyang Wu", + "Jingxi Chen", + "Jing Xie", + "Xiaomin Wu", + "Lichao Sun" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2607.01709", + "source": "arxiv", + "source_id": "arxiv:2607.01709", + "pdf_url": "https://arxiv.org/pdf/2607.01709", + "primary_query": "agent-memory" + }, + { + "id": "2607.02807", + "title": "SwarmResearch: Orchestrating Coding Agents for Open-Ended Discovery", + "url": "https://arxiv.org/abs/2607.02807", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Yuvraj Virk", + "Zack Edds", + "Chunqiu Steven Xia", + "Lingming Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "coding-agent", + "computer-use", + "multi-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.02807", + "source": "arxiv", + "source_id": "arxiv:2607.02807", + "pdf_url": "https://arxiv.org/pdf/2607.02807", + "primary_query": "coding-agent" + }, + { + "id": "2607.02448", + "title": "AgentsCAD: Automated Design for Manufacturing of FDM Parts via Multi-Agent LLM Reasoning and Geometric Feature Recognition", + "url": "https://arxiv.org/abs/2607.02448", + "published": "2026-07-02", + "updated": "2026-07-07", + "authors": [ + "Emmanuel George", + "Christopher Keefe", + "Peter Pak", + "Amir Barati Farimani" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.02448", + "source": "arxiv", + "source_id": "arxiv:2607.02448", + "pdf_url": "https://arxiv.org/pdf/2607.02448", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.01213", + "title": "RepoRescue: An Empirical Study of LLM Agents on Whole-Repository Compatibility Rescue", + "url": "https://arxiv.org/abs/2607.01213", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Zhihao Lin", + "Mingyi Zhou", + "Zhensu Sun", + "Yizhuo Yang", + "Renyu Yang", + "David Lo", + "Li Li" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01213", + "source": "arxiv", + "source_id": "arxiv:2607.01213", + "pdf_url": "https://arxiv.org/pdf/2607.01213", + "primary_query": "llm-agent" + }, + { + "id": "2607.01120", + "title": "Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents", + "url": "https://arxiv.org/abs/2607.01120", + "published": "2026-07-01", + "updated": "2026-07-02", + "authors": [ + "Ran Yan", + "Wei Fu", + "Jiale Li", + "Shusheng Xu", + "Zhiyu Mei", + "Jiaxuan Gao", + "Jiarui Zhang", + "Wentai Zhang", + "Hao Dai", + "Xujie Shen", + "Chuyi He", + "Zhen Pu", + "Jun Mei", + "Zhiyao Lin", + "Haitao Wang", + "Zhiqiang Ding", + "Jiawei Zhang", + "Huaijie Wang", + "Ruida Xu", + "Honghua Dong", + "Youhe Jiang", + "Yi Wu", + "Tongkai Yang", + "Binhang Yuan" + ], + "categories": [ + "cs.DC" + ], + "topics": [ + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01120", + "source": "arxiv", + "source_id": "arxiv:2607.01120", + "pdf_url": "https://arxiv.org/pdf/2607.01120", + "primary_query": "llm-agent" + }, + { + "id": "2607.00345", + "title": "Registry-Governed Agent Lifecycle:Completing EDDOps with Evaluation-DrivenRegistration, Promotion, and Retirement on AWS AgentCore", + "url": "https://arxiv.org/abs/2607.00345", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Richard Kang", + "Vincent Wang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00345", + "source": "arxiv", + "source_id": "arxiv:2607.00345", + "pdf_url": "https://arxiv.org/pdf/2607.00345", + "primary_query": "llm-agent" + }, + { + "id": "2607.00911", + "title": "From Registry to Repository: How AI Agent Skills Are Written, Adapted, and Maintained", + "url": "https://arxiv.org/abs/2607.00911", + "published": "2026-07-01", + "updated": "2026-07-06", + "authors": [ + "Haoyu Gao", + "Jai Lal Lulla", + "Hong Yi Lin", + "Sebastian Baltes", + "Christoph Treude", + "Mansooreh Zahedi" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "coding-agent", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2607.00911", + "source": "arxiv", + "source_id": "arxiv:2607.00911", + "pdf_url": "https://arxiv.org/pdf/2607.00911", + "primary_query": "ai-agent" + }, + { + "id": "2607.01421", + "title": "Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance", + "url": "https://arxiv.org/abs/2607.01421", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Laxmipriya Ganesh Iyer" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.01421", + "source": "arxiv", + "source_id": "arxiv:2607.01421", + "pdf_url": "https://arxiv.org/pdf/2607.01421", + "primary_query": "agentic-ai" + }, + { + "id": "2607.01061", + "title": "Agentic generation of verifiable rules for deterministic, self-expanding reaction classification", + "url": "https://arxiv.org/abs/2607.01061", + "published": "2026-07-01", + "updated": "2026-07-05", + "authors": [ + "Daniel Armstrong", + "Maarten Dobbelaere", + "Valentas Olikauskas", + "Helena Avila", + "Octavian Susanu", + "Jérôme Waser", + "Philippe Schwaller" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "coding-agent", + "multi-agent", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.01061", + "source": "arxiv", + "source_id": "arxiv:2607.01061", + "pdf_url": "https://arxiv.org/pdf/2607.01061", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31744", + "title": "A Conversational Agentic Interface to Physics-Based Household Digital Twins for Residential Energy Decision Support", + "url": "https://arxiv.org/abs/2606.31744", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Costas Mylonas", + "Titos Georgoulakis", + "Magda Foti" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.31744", + "source": "arxiv", + "source_id": "arxiv:2606.31744", + "pdf_url": "https://arxiv.org/pdf/2606.31744", + "primary_query": "llm-agent" + }, + { + "id": "2606.31209", + "title": "Long-term Traffic Simulation via Structured Autoregressive Modeling", + "url": "https://arxiv.org/abs/2606.31209", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Lingyu Xiao", + "Zexin Feng", + "Xintao Yan" + ], + "categories": [ + "cs.AI", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31209", + "source": "arxiv", + "source_id": "arxiv:2606.31209", + "pdf_url": "https://arxiv.org/pdf/2606.31209", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.31085", + "title": "DDIAgents: Mechanism-Conditioned Context Flow for Drug-Drug Interaction Prediction", + "url": "https://arxiv.org/abs/2606.31085", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Zhenqian Shen", + "Yu Liu", + "Xiaoyi Fu", + "Quanming Yao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31085", + "source": "arxiv", + "source_id": "arxiv:2606.31085", + "pdf_url": "https://arxiv.org/pdf/2606.31085", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.30383", + "title": "Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents", + "url": "https://arxiv.org/abs/2606.30383", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Bojie Li", + "Noah Shi" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.30383", + "source": "arxiv", + "source_id": "arxiv:2606.30383", + "pdf_url": "https://arxiv.org/pdf/2606.30383", + "primary_query": "llm-agent" + }, + { + "id": "2606.30697", + "title": "LUMOS: A Semantic Operating-System Layer for Accessibility-Grounded AI Agents", + "url": "https://arxiv.org/abs/2606.30697", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yogeswar Reddy Thota" + ], + "categories": [ + "cs.OS", + "cs.AI", + "cs.CV" + ], + "topics": [ + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "web-gui-agent" + ], + "arxiv_id": "2606.30697", + "source": "arxiv", + "source_id": "arxiv:2606.30697", + "pdf_url": "https://arxiv.org/pdf/2606.30697", + "primary_query": "ai-agent" + }, + { + "id": "2606.30877", + "title": "A Systematic Approach to Multi-Agent AI from Advanced Regulatory Control Theory: Safe and Auditable LLM Operator Agents for Process Control", + "url": "https://arxiv.org/abs/2606.30877", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Idelfonso B. R. Nogueira", + "Sigurd Skogestad" + ], + "categories": [ + "eess.SY", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.30877", + "source": "arxiv", + "source_id": "arxiv:2606.30877", + "pdf_url": "https://arxiv.org/pdf/2606.30877", + "primary_query": "agentic-ai" + }, + { + "id": "2606.29957", + "title": "SWE-Together: Evaluating Coding Agents in Interactive User Sessions", + "url": "https://arxiv.org/abs/2606.29957", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yifan Wu", + "Zhuokai Zhao", + "Songlin Li", + "Ho Hin Lee", + "Jiacheng Zhu", + "Shirley Wu", + "Tianhe Yu", + "Serena Li", + "Lizhu Zhang", + "Xiangjun Fan", + "Shengzhi Li" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "coding-agent" + ], + "arxiv_id": "2606.29957", + "source": "arxiv", + "source_id": "arxiv:2606.29957", + "pdf_url": "https://arxiv.org/pdf/2606.29957", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.29778", + "title": "Mandol: An Agglomerative Agent Memory System for Long-Term Conversations", + "url": "https://arxiv.org/abs/2606.29778", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yuhan Zhang", + "Zhiyuan Guo", + "Ziheng Zeng", + "Wei Wang", + "Wentao Wu", + "Lijie Xu" + ], + "categories": [ + "cs.DB", + "cs.AI", + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "rag-agent" + ], + "arxiv_id": "2606.29778", + "source": "arxiv", + "source_id": "arxiv:2606.29778", + "pdf_url": "https://arxiv.org/pdf/2606.29778", + "primary_query": "agent-memory" + }, + { + "id": "2606.30573", + "title": "SWE-INTERACT: Reimagining SWE Benchmarks as User-Driven Long-Horizon Coding Sessions", + "url": "https://arxiv.org/abs/2606.30573", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Mohit Raghavendra", + "Anisha Gunjal", + "Aakash Sabharwal", + "Yunzhong He" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.30573", + "source": "arxiv", + "source_id": "arxiv:2606.30573", + "pdf_url": "https://arxiv.org/pdf/2606.30573", + "primary_query": "coding-agent" + }, + { + "id": "2606.30560", + "title": "TraceLab: Characterizing Coding Agent Workloads for LLM Serving", + "url": "https://arxiv.org/abs/2606.30560", + "published": "2026-06-29", + "updated": "2026-06-30", + "authors": [ + "Kan Zhu", + "Mathew Jacob", + "Chenxi Ma", + "Yi Pan", + "Stephanie Wang", + "Arvind Krishnamurthy", + "Baris Kasikci" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.PF" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.30560", + "source": "arxiv", + "source_id": "arxiv:2606.30560", + "pdf_url": "https://arxiv.org/pdf/2606.30560", + "primary_query": "coding-agent" + }, + { + "id": "2606.30887", + "title": "Training Therapeutic Judges and Multi-Agent Systems for Human-Aligned Mental Health Support", + "url": "https://arxiv.org/abs/2606.30887", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Mizanur Rahman", + "Abeer Badawi", + "Elahe Rahimi", + "Laleh Seyyed-Kalantari", + "Frank Rudzicz", + "Enamul Hoque", + "Elham Dolatabadi" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30887", + "source": "arxiv", + "source_id": "arxiv:2606.30887", + "pdf_url": "https://arxiv.org/pdf/2606.30887", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00038", + "title": "Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting", + "url": "https://arxiv.org/abs/2607.00038", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Sandeco Macedo" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.00038", + "source": "arxiv", + "source_id": "arxiv:2607.00038", + "pdf_url": "https://arxiv.org/pdf/2607.00038", + "primary_query": "coding-agent" + }, + { + "id": "2606.29014", + "title": "Customized Generative AI Agent for Transportation Engineering Practice: A Development and Continued Pre-training Guideline", + "url": "https://arxiv.org/abs/2606.29014", + "published": "2026-06-27", + "updated": "2026-07-03", + "authors": [ + "Dianwei Chen", + "Yuan-Zheng Lei", + "Zifan Zhang", + "Yuchen Liu", + "Xianfeng Yang" + ], + "categories": [ + "cs.AI", + "cs.DL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.29014", + "source": "arxiv", + "source_id": "arxiv:2606.29014", + "pdf_url": "https://arxiv.org/pdf/2606.29014", + "primary_query": "ai-agent" + }, + { + "id": "2606.28781", + "title": "HyphaeDB: A Living Knowledge Topology for Agent-First Memory", + "url": "https://arxiv.org/abs/2606.28781", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Krishna Halaharvi" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "coding-agent", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "agentic-ai" + ], + "arxiv_id": "2606.28781", + "source": "arxiv", + "source_id": "arxiv:2606.28781", + "pdf_url": "https://arxiv.org/pdf/2606.28781", + "primary_query": "agent-memory" + }, + { + "id": "2606.28839", + "title": "The Contagion Tensor: A Framework for Measuring Output-Distribution Coupling in Multi-Agent LLM Systems -- and Auditing the Claims It Enables", + "url": "https://arxiv.org/abs/2606.28839", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Zewen Liu" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28839", + "source": "arxiv", + "source_id": "arxiv:2606.28839", + "pdf_url": "https://arxiv.org/pdf/2606.28839", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28270", + "title": "Agent-Native Immune System: Architecture, Taxonomy, and Engineering", + "url": "https://arxiv.org/abs/2606.28270", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Bo Shen", + "Lifeng Chang", + "Tianyuan Wei", + "Yunpeng Li", + "Feng Shi", + "Yichen Han", + "Peijie Gao", + "Shiyi Kuang", + "Xin Chang", + "Dehui Li" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.28270", + "source": "arxiv", + "source_id": "arxiv:2606.28270", + "pdf_url": "https://arxiv.org/pdf/2606.28270", + "primary_query": "tool-use" + }, + { + "id": "2606.28436", + "title": "Dockerless: Environment-Free Program Verifier for Coding Agents", + "url": "https://arxiv.org/abs/2606.28436", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Wenhao Zeng", + "Yuling Shi", + "Xiaodong Gu", + "Chao Hu", + "Chaofan Wang", + "Yuhao Cui", + "Hongting Zhou", + "Mengnan Qi", + "Jianqiao Wangni", + "Zhaojian Yu", + "Shuzheng Gao", + "Kai Cai", + "Shilin He" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.28436", + "source": "arxiv", + "source_id": "arxiv:2606.28436", + "pdf_url": "https://arxiv.org/pdf/2606.28436", + "primary_query": "coding-agent" + }, + { + "id": "2606.26918", + "title": "Diagnosing Task Insensitivity in Language Agents", + "url": "https://arxiv.org/abs/2606.26918", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Jingyu Liu", + "Xiaopeng Wu", + "Kehan Chen", + "Chuan Yu", + "Yong Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.26918", + "source": "arxiv", + "source_id": "arxiv:2606.26918", + "pdf_url": "https://arxiv.org/pdf/2606.26918", + "primary_query": "language-agent" + }, + { + "id": "2606.26524", + "title": "VIGIL: Runtime Enforcement of Behavioral Specifications in AI Agent Skills", + "url": "https://arxiv.org/abs/2606.26524", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Ying Li", + "Yanju Chen", + "Hongbo Wen", + "Bosi Zhang", + "Hanzhi Liu", + "Peiran Wang", + "Yu Feng", + "Yuan Tian" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.26524", + "source": "arxiv", + "source_id": "arxiv:2606.26524", + "pdf_url": "https://arxiv.org/pdf/2606.26524", + "primary_query": "ai-agent" + }, + { + "id": "2606.27499", + "title": "DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection", + "url": "https://arxiv.org/abs/2606.27499", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Yujin Tang", + "Chenming Shang", + "Ruize Xu", + "Nikhil Singh" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.27499", + "source": "arxiv", + "source_id": "arxiv:2606.27499", + "pdf_url": "https://arxiv.org/pdf/2606.27499", + "primary_query": "agent-memory" + }, + { + "id": "2606.27243", + "title": "NOVA: A Verification-Aware Agent Harness for Architecture Evolution in Industrial Recommender Systems", + "url": "https://arxiv.org/abs/2606.27243", + "published": "2026-06-25", + "updated": "2026-06-26", + "authors": [ + "Shaohua Liu", + "Liang Fang", + "Yilong Sun", + "Shudong Huang", + "Qingsong Luo", + "Shaoxin Liu", + "Xiaoyang Chen", + "Dongqiang Liu", + "Chuangang Ma", + "Zhenzhen Chai", + "Henghuan Wang", + "Shijie Quan", + "Changyuan Cui", + "Zhangbin Zhu", + "Peng Chen", + "Wei Xu", + "Lei Xiao", + "Haijie Gu", + "Jie Jiang" + ], + "categories": [ + "cs.IR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "memory", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.27243", + "source": "arxiv", + "source_id": "arxiv:2606.27243", + "pdf_url": "https://arxiv.org/pdf/2606.27243", + "primary_query": "coding-agent" + }, + { + "id": "2606.26924", + "title": "A Deterministic Control Plane for LLM Coding Agents", + "url": "https://arxiv.org/abs/2606.26924", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Padmaraj Madatha" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.26924", + "source": "arxiv", + "source_id": "arxiv:2606.26924", + "pdf_url": "https://arxiv.org/pdf/2606.26924", + "primary_query": "coding-agent" + }, + { + "id": "2606.26883", + "title": "EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents", + "url": "https://arxiv.org/abs/2606.26883", + "published": "2026-06-25", + "updated": "2026-06-30", + "authors": [ + "Ryuji Hashimoto", + "Masahiro Kaneko", + "Kentaro Ueda", + "Takehiro Takayanagi", + "Kiyoshi Izumi" + ], + "categories": [ + "cs.DL" + ], + "topics": [ + "memory", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.26883", + "source": "arxiv", + "source_id": "arxiv:2606.26883", + "pdf_url": "https://arxiv.org/pdf/2606.26883", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28409", + "title": "Evidence-Driven LLM Agent for C-to-Synthesizable-C Conversion and Verification", + "url": "https://arxiv.org/abs/2606.28409", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Zhe Zhao", + "Hongbing Lang", + "Zhihan Xiao", + "Luke Ztz Hu", + "John Imoleayo Adebisi", + "Songping Mai" + ], + "categories": [ + "cs.AR", + "cs.AI" + ], + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.28409", + "source": "arxiv", + "source_id": "arxiv:2606.28409", + "pdf_url": "https://arxiv.org/pdf/2606.28409", + "primary_query": "rag-agent" + }, + { + "id": "2606.25819", + "title": "Beyond Function Calling: Benchmarking Tool-Using Agents under Tool-Environment Unreliability", + "url": "https://arxiv.org/abs/2606.25819", + "published": "2026-06-24", + "updated": "2026-06-27", + "authors": [ + "Yang Tian", + "Zhengpeng Shi", + "Yu Zhou", + "Bo Zhao" + ], + "categories": [ + "cs.CL", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling", + "tool-use" + ], + "arxiv_id": "2606.25819", + "source": "arxiv", + "source_id": "arxiv:2606.25819", + "pdf_url": "https://arxiv.org/pdf/2606.25819", + "primary_query": "function-calling" + }, + { + "id": "2606.26453", + "title": "Optimizing CUDA like a Human: Micro-Profiling Tools as Expert Surrogates for LLM-Based GPU Kernel Optimization", + "url": "https://arxiv.org/abs/2606.26453", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Jiading Gai", + "Shuai Zhang", + "Kaj Bostrom", + "Jin Huang", + "Vihang Patil", + "Haoyang Fang", + "Bernie Wang", + "Huzefa Rangwala", + "George Karypis" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "multi-agent-llm" + ], + "arxiv_id": "2606.26453", + "source": "arxiv", + "source_id": "arxiv:2606.26453", + "pdf_url": "https://arxiv.org/pdf/2606.26453", + "primary_query": "coding-agent" + }, + { + "id": "2606.25361", + "title": "Memory Makes the Difference: Evaluating How Different Memory Roles Shape Conversational Agents", + "url": "https://arxiv.org/abs/2606.25361", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yuxin Wang", + "Paul Thomas", + "Zhiwei Yu", + "Yuan Gao", + "Saeed Hassanpour", + "Soroush Vosoughi", + "Robert Sim", + "Nick Craswell" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.25361", + "source": "arxiv", + "source_id": "arxiv:2606.25361", + "pdf_url": "https://arxiv.org/pdf/2606.25361", + "primary_query": "rag-agent" + }, + { + "id": "2606.27397", + "title": "SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game", + "url": "https://arxiv.org/abs/2606.27397", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Yeqi Feng", + "Yuxin Chen", + "Tianxing He" + ], + "categories": [ + "cs.MA", + "cs.AI", + "cs.GT" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.27397", + "source": "arxiv", + "source_id": "arxiv:2606.27397", + "pdf_url": "https://arxiv.org/pdf/2606.27397", + "primary_query": "planning-agent" + }, + { + "id": "2606.25139", + "title": "Buildrix: An Open Platform for Sharing and Benchmarking Agentic AI Skills in Building Engineering", + "url": "https://arxiv.org/abs/2606.25139", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Zixin Jiang", + "Bing Dong" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.25139", + "source": "arxiv", + "source_id": "arxiv:2606.25139", + "pdf_url": "https://arxiv.org/pdf/2606.25139", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24779", + "title": "DeepBD: A Grounded Agentic Workflow for Variant Prioritization and Diagnosis of Genetic Birth Defects", + "url": "https://arxiv.org/abs/2606.24779", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Shiyu Li", + "Ziqi Yan", + "Zhihao Wu", + "Jielong Lu", + "Weiran Liao", + "Jiajun Yu", + "Genjie Li", + "Zeyu Chu", + "Jiajun Bu", + "Haishuai Wang" + ], + "categories": [ + "q-bio.GN", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.24779", + "source": "arxiv", + "source_id": "arxiv:2606.24779", + "pdf_url": "https://arxiv.org/pdf/2606.24779", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24193", + "title": "SkyChain Intelligence: A Blockchain-Secured Multi-Agent DRL Framework for Low-Altitude Embodied Artificial Intelligence", + "url": "https://arxiv.org/abs/2606.24193", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Haoxiang Luo", + "Tianqi Jiang", + "Ruichen Zhang", + "Yinqiu Liu", + "Gang Sun", + "Hongfang Yu", + "Abbas Jamalipour", + "Dong In Kim" + ], + "categories": [ + "cs.NI", + "cs.DC" + ], + "topics": [ + "agent-safety", + "embodied-agent", + "multi-agent", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.24193", + "source": "arxiv", + "source_id": "arxiv:2606.24193", + "pdf_url": "https://arxiv.org/pdf/2606.24193", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24597", + "title": "Qwen-AgentWorld: Language World Models for General Agents", + "url": "https://arxiv.org/abs/2606.24597", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yuxin Zuo", + "Zikai Xiao", + "Li Sheng", + "Fei Huang", + "Jianhong Tu", + "Yuxuan Liu", + "Tianyi Tang", + "Xiaomeng Hu", + "Yang Su", + "Qingfeng Lan", + "Yantao Liu", + "Qin Zhu", + "Yinger Zhang", + "Bowen Yu", + "Haiquan Zhao", + "Haiyang Xu", + "Jianxin Yang", + "Jiayang Cheng", + "Junyang Wang", + "Lianghao Deng", + "Mingfeng Xue", + "Tianyi Bai", + "Yang Fan", + "Yubo Ma", + "Yucheng Li", + "Zeyu Cui", + "Zhihai Wang", + "Zhihui Xie", + "Zhuorui Ye", + "An Yang", + "Dayiheng Liu", + "Jingren Zhou", + "Ning Ding" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.24597", + "source": "arxiv", + "source_id": "arxiv:2606.24597", + "pdf_url": "https://arxiv.org/pdf/2606.24597", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.25115", + "title": "Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory", + "url": "https://arxiv.org/abs/2606.25115", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Beining Wu", + "Zihao Ding", + "Jun Huang", + "Yanxiao Zhao" + ], + "categories": [ + "cs.LG", + "cs.NI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.25115", + "source": "arxiv", + "source_id": "arxiv:2606.25115", + "pdf_url": "https://arxiv.org/pdf/2606.25115", + "primary_query": "agent-memory" + }, + { + "id": "2606.24322", + "title": "Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees", + "url": "https://arxiv.org/abs/2606.24322", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yedidel Louck" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.24322", + "source": "arxiv", + "source_id": "arxiv:2606.24322", + "pdf_url": "https://arxiv.org/pdf/2606.24322", + "primary_query": "agent-memory" + }, + { + "id": "2606.24839", + "title": "Grading the Grader: Lessons from Evaluating an Agentic Data Analysis System", + "url": "https://arxiv.org/abs/2606.24839", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Tian Zheng", + "Kai-Tai Hsu" + ], + "categories": [ + "cs.AI", + "stat.AP" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24839", + "source": "arxiv", + "source_id": "arxiv:2606.24839", + "pdf_url": "https://arxiv.org/pdf/2606.24839", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.23927", + "title": "RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems", + "url": "https://arxiv.org/abs/2606.23927", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Yarin Yerushalmi Levi", + "Roy Betser", + "Amit Giloni", + "Lidor Erez", + "Itay Gershon", + "Oren Rachmil", + "Sindhu Padakandla", + "Roman Vainshtein" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.23927", + "source": "arxiv", + "source_id": "arxiv:2606.23927", + "pdf_url": "https://arxiv.org/pdf/2606.23927", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24551", + "title": "GUI vs. CLI: Execution Bottlenecks in Screen-Only and Skill-Mediated Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.24551", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Xiao Zhou", + "Siyue Zhang", + "Yilun Zhao", + "Jinbiao Wei", + "Tingyu Song", + "Arman Cohan", + "Chen Zhao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.24551", + "source": "arxiv", + "source_id": "arxiv:2606.24551", + "pdf_url": "https://arxiv.org/pdf/2606.24551", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.22948", + "title": "ENVS: Environment-Native Verified Search for Long-Horizon GUI Agents", + "url": "https://arxiv.org/abs/2606.22948", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Yincheng Zhou", + "Athena Zhuoming Zhong", + "Shijie Zhang", + "Kevin Zhang", + "Teresa Xiaotao Shang", + "Shanghang Zhang" + ], + "categories": [ + "cs.AI", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.22948", + "source": "arxiv", + "source_id": "arxiv:2606.22948", + "pdf_url": "https://arxiv.org/pdf/2606.22948", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.23764", + "title": "Emergent Relational Order in LLM Agent Societies: From Collective Affect to Authority Stratification", + "url": "https://arxiv.org/abs/2606.23764", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Zhiyuan Ji", + "Xinyu Chen", + "Ziqi Dai", + "Shiyun Tang", + "Chunyu Wei", + "Yueguo Chen" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.23764", + "source": "arxiv", + "source_id": "arxiv:2606.23764", + "pdf_url": "https://arxiv.org/pdf/2606.23764", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.23343", + "title": "Group Selection Promotes Prosocial Prompts in Populations of LLM Agents", + "url": "https://arxiv.org/abs/2606.23343", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Luis Celiktemel", + "Edward Eichhorn", + "Levin Brinkmann", + "Robin Schimmelpfennig", + "Aron Vallinder", + "Yaomin Jiang", + "Edward Hughes", + "Iyad Rahwan" + ], + "categories": [ + "cs.CY" + ], + "topics": [ + "coding-agent", + "computer-use", + "multi-agent", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.23343", + "source": "arxiv", + "source_id": "arxiv:2606.23343", + "pdf_url": "https://arxiv.org/pdf/2606.23343", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.22388", + "title": "PlanBench-XL: Evaluating Long-Horizon Planning of LLM Tool-Use Agents in Large-Scale Tool Ecosystems", + "url": "https://arxiv.org/abs/2606.22388", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Jiayu Liu", + "Qihan Lin", + "Cheng Qian", + "Rui Wang", + "Emre Can Acikgoz", + "Xiaocheng Yang", + "Jiateng Liu", + "Zhenhailong Wang", + "Xiusi Chen", + "Heng Ji", + "Dilek Hakkani-Tür" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent", + "tool-use" + ], + "arxiv_id": "2606.22388", + "source": "arxiv", + "source_id": "arxiv:2606.22388", + "pdf_url": "https://arxiv.org/pdf/2606.22388", + "primary_query": "planning-agent" + }, + { + "id": "2606.22610", + "title": "PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement", + "url": "https://arxiv.org/abs/2606.22610", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Weiwei Ye", + "Hangchen Liu", + "Dongyuan Li", + "Renhe Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.22610", + "source": "arxiv", + "source_id": "arxiv:2606.22610", + "pdf_url": "https://arxiv.org/pdf/2606.22610", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.21565", + "title": "Composing Verifiable Conceptual Models via Building Blocks: Towards Design-Time Verification of Agentic AI Workflows", + "url": "https://arxiv.org/abs/2606.21565", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Noe Y. Flandre", + "Alexander C. Nwala", + "Philippe J. Giabbanelli" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.21565", + "source": "arxiv", + "source_id": "arxiv:2606.21565", + "pdf_url": "https://arxiv.org/pdf/2606.21565", + "primary_query": "agentic-ai" + }, + { + "id": "2606.21732", + "title": "Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents", + "url": "https://arxiv.org/abs/2606.21732", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Zesen Liu", + "Zihan Zhang", + "Dongdong She" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.21732", + "source": "arxiv", + "source_id": "arxiv:2606.21732", + "pdf_url": "https://arxiv.org/pdf/2606.21732", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.20479", + "title": "GroundControl: Anticipating Navigation Failures in Vision-Language Agents via Trajectory-Consistent Uncertainty Estimates", + "url": "https://arxiv.org/abs/2606.20479", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Nastaran Darabi", + "Divake Kumar", + "Sina Tayebati", + "Devashri Naik", + "Amit Ranjan Trivedi" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.20479", + "source": "arxiv", + "source_id": "arxiv:2606.20479", + "pdf_url": "https://arxiv.org/pdf/2606.20479", + "primary_query": "language-agent" + }, + { + "id": "2606.20041", + "title": "AI Economist Agent: An Agentic Framework for Model-Grounded Economic Analysis with RAG, Knowledge Graphs, and Large Language Models", + "url": "https://arxiv.org/abs/2606.20041", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Masahiro Kato" + ], + "categories": [ + "econ.GN", + "cs.AI", + "cs.LG", + "q-fin.GN" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2606.20041", + "source": "arxiv", + "source_id": "arxiv:2606.20041", + "pdf_url": "https://arxiv.org/pdf/2606.20041", + "primary_query": "ai-agent" + }, + { + "id": "2606.19852", + "title": "Prompt, Plan, Extract: Zero-Shot Agentic LLMs Workflows for Lung Pathology Extraction from Clinical Narratives", + "url": "https://arxiv.org/abs/2606.19852", + "published": "2026-06-18", + "updated": "2026-06-25", + "authors": [ + "Aman Pathak", + "Cheng Peng", + "Mengxian Lyu", + "Ziyi Chen", + "Reema Solan", + "Sankalp Talankar", + "Yasir Khan", + "Hiren Mehta", + "Aokun Chen", + "Yi Guo", + "Yonghui Wu" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.19852", + "source": "arxiv", + "source_id": "arxiv:2606.19852", + "pdf_url": "https://arxiv.org/pdf/2606.19852", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20047", + "title": "PACMS: Submodular Context Selection as a Pluggable Engine for LLM Agents", + "url": "https://arxiv.org/abs/2606.20047", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Manu Ghulyani", + "Arunabh Singh", + "Karan Bharadwaj", + "Ankit Nath", + "Suranjan Goswami" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.20047", + "source": "arxiv", + "source_id": "arxiv:2606.20047", + "pdf_url": "https://arxiv.org/pdf/2606.20047", + "primary_query": "tool-use" + }, + { + "id": "2606.19926", + "title": "MemGUI-Agent: An End-to-End Long-Horizon Mobile GUI Agent with Proactive Context Management", + "url": "https://arxiv.org/abs/2606.19926", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Guangyi Liu", + "Gao Wu", + "Congxiao Liu", + "Pengxiang Zhao", + "Liang Liu", + "Mading Li", + "Qi Zhang", + "Mengyan Wang", + "Liang Guo", + "Yong Liu" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.19926", + "source": "arxiv", + "source_id": "arxiv:2606.19926", + "pdf_url": "https://arxiv.org/pdf/2606.19926", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.20922", + "title": "Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning", + "url": "https://arxiv.org/abs/2606.20922", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Shanghao Shi", + "Xiao Wang", + "Chaoyu Zhang", + "Hao Li", + "Wenjing Lou", + "Thomas Hou", + "Yevgeniy Vorobeychik", + "Chongjie Zhang", + "Ning Zhang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.20922", + "source": "arxiv", + "source_id": "arxiv:2606.20922", + "pdf_url": "https://arxiv.org/pdf/2606.20922", + "primary_query": "planning-agent" + }, + { + "id": "2606.19787", + "title": "ORAgentBench: Can LLM Agents Solve Challenging Operations Research Tasks End to End?", + "url": "https://arxiv.org/abs/2606.19787", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Jiajun Li", + "Mingshu Cai", + "Yixuan Li", + "Yu Ding", + "Ran Hou", + "Guanyu Nie", + "Xiongwei Han", + "Wanyuan Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.19787", + "source": "arxiv", + "source_id": "arxiv:2606.19787", + "pdf_url": "https://arxiv.org/pdf/2606.19787", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.18068", + "title": "Agentic AI-based Framework for Mitigating Premature Diagnostic Handoff and Silent Hallucination in Healthcare Applications", + "url": "https://arxiv.org/abs/2606.18068", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Divyansh Srivastava", + "Shreya Ghosh", + "Anshul Verma", + "Rajkumar Buyya" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.18068", + "source": "arxiv", + "source_id": "arxiv:2606.18068", + "pdf_url": "https://arxiv.org/pdf/2606.18068", + "primary_query": "agentic-ai" + }, + { + "id": "2606.17680", + "title": "EnvRL: Learn from Environment Dynamics in Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.17680", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Zhitong Wang", + "Songze Li", + "Hao Peng", + "Shuzheng Si", + "Yi Wang", + "Maosong Sun", + "Juanzi Li" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.17680", + "source": "arxiv", + "source_id": "arxiv:2606.17680", + "pdf_url": "https://arxiv.org/pdf/2606.17680", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18406", + "title": "CoreMem: Riemannian Retrieval and Fisher-Guided Distillation for Long-Term Memory in Dialogue Agents", + "url": "https://arxiv.org/abs/2606.18406", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Jiaqi Chen", + "Yongqin Zeng", + "Shaoshen Chen", + "Yijian Zhang", + "Hai-Tao Zheng", + "Chunxia Ma", + "XiuTeng Zhou" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.18406", + "source": "arxiv", + "source_id": "arxiv:2606.18406", + "pdf_url": "https://arxiv.org/pdf/2606.18406", + "primary_query": "agent-memory" + }, + { + "id": "2606.17449", + "title": "MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation", + "url": "https://arxiv.org/abs/2606.17449", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Zehang Wei", + "Jiaxin Dai", + "Jiamin Yan", + "Xiang Xiang" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.CV", + "cs.LG", + "cs.MM" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.17449", + "source": "arxiv", + "source_id": "arxiv:2606.17449", + "pdf_url": "https://arxiv.org/pdf/2606.17449", + "primary_query": "rag-agent" + }, + { + "id": "2606.16432", + "title": "ACCORD: Action-Conditioned Contextual Grounding for Language Agents", + "url": "https://arxiv.org/abs/2606.16432", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Lai Jiang", + "Cheng Qian", + "Zhenhailong Wang", + "Pan Lu", + "Heng Ji", + "Hao Peng" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.16432", + "source": "arxiv", + "source_id": "arxiv:2606.16432", + "pdf_url": "https://arxiv.org/pdf/2606.16432", + "primary_query": "language-agent" + }, + { + "id": "2606.15903", + "title": "Control-Plane Placement Shapes Forgetting: An Architectural Study of Agent Memory Across Thirteen System Configurations", + "url": "https://arxiv.org/abs/2606.15903", + "published": "2026-06-14", + "updated": "2026-06-16", + "authors": [ + "Dongxu Yang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.15903", + "source": "arxiv", + "source_id": "arxiv:2606.15903", + "pdf_url": "https://arxiv.org/pdf/2606.15903", + "primary_query": "agent-memory" + }, + { + "id": "2606.15609", + "title": "FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion", + "url": "https://arxiv.org/abs/2606.15609", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Zixin Rao", + "Wentian Zhu", + "Chan Aristella Lu", + "Zhaorun Chen", + "Wei Niu", + "Le Guan", + "Bo Li", + "Zhen Xiang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.15609", + "source": "arxiv", + "source_id": "arxiv:2606.15609", + "pdf_url": "https://arxiv.org/pdf/2606.15609", + "primary_query": "agent-memory" + }, + { + "id": "2606.15079", + "title": "Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale", + "url": "https://arxiv.org/abs/2606.15079", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Ang Li", + "Ben Liu", + "Bin Han", + "Bin Hu", + "Bin Jing", + "Binbin Hu", + "Bing Li", + "Cai Chen", + "Caizhi Tang", + "Changxin Tian", + "Chao Huang", + "Chao Zhang", + "Chen Liang", + "Chen Qian", + "Chengfu Tang", + "Chengyao Wen", + "Chilin Fu", + "Chunwei Wu", + "Cong Zhang", + "Cunyin Peng", + "Daixin Wang", + "Dalong Zhang", + "Deng Zhao", + "Dingnan Jin", + "Dingyuan Zhu", + "Donghao Zhang", + "Fan Yuan", + "Fangzheng Zhao", + "Fanzhuang Meng", + "Feifan Wu", + "Feng Xu", + "Fengbin Fang", + "Gangshan Wang", + "Guodong Yang", + "Hailin Zhao", + "Haitao Wang", + "Haitao Zhang", + "Hanxiao Zhang", + "Hanzi Wang", + "Hao Dai", + "Hao Liu", + "Hao Qian", + "Hao Wu", + "Haoxiong Liu", + "Haoyu Xu", + "Heng Zhang", + "Hong Liu", + "Hongliang Zhang", + "Hongrui Liu", + "Hongxun Li", + "Hongzhi Ruan", + "Huaidong Xiong", + "Huihuang Zheng", + "Huikang Tang", + "Jia Guo", + "Jia Li", + "Jia Liu", + "Jiameng Wang", + "Jiaming Liu", + "Jiannan Shi", + "Jianping Wei", + "Jiaolong Yang", + "Jiapeng Wang", + "Jie Gao", + "Jie Wang", + "Jiewei Wu", + "Jin Yang", + "Jinjin Li", + "Jinjing Huang", + "Jinquan Sun", + "Jinyao Chen", + "Juanhui Tu", + "Jun Liu", + "Jun Mei", + "Jun Xu", + "Jun Zhou", + "Junjie Ou", + "Junnan Sipan", + "Junpeng Fang", + "Kaihong Zhang", + "Kaiqin Hu", + "Ke Shi", + "Kuan Xu", + "Kun Tang", + "Kunlong Chen", + "Lanyin Mei", + "Lei Chen", + "Lei Liang", + "Lei Xu", + "Li Tang", + "Liang Jiang", + "Liangcheng Fu", + "Lihui Zhang", + "Linfeng Shi", + "Lintao Ma", + "Liyuan Liu", + "Longfei Li", + "Longfei Zheng", + "Lu Liu", + "Lu Yu", + "Man Li", + "Meiqi Zhu", + "Meng Li", + "Mengjie Gao", + "Mengshu Sun", + "Mingming Yin", + "Mingyang Zhang", + "Mingyuan Fan", + "Nuo Xu", + "Pan Tang", + "Peijie Jiang", + "Peilong Zhao", + "Peng Lin", + "Pingping Liu", + "Qi Zuo", + "Qian Zhao", + "Qiang Cheng", + "Qianggang Cao", + "Qiaoben Bao", + "Qing Cui", + "Qingyuan Yang", + "Qitao Shi", + "Qiyin Huang", + "Qizheng Zhou", + "Quan Wan", + "Runyuan Zhao", + "Shaomian Zheng", + "Shaowei Wei", + "Shengnan Zhang", + "Shuaicheng Li", + "Shujie Li", + "Shuo Zhang", + "Sikang Bian", + "Tianchu Yao", + "Tiange Xu", + "Tianshu Wang", + "Ting Guo", + "Tinghao Wang", + "Tingwei Huang", + "Tong Zhao", + "Tongkai Yang", + "Wang Hong", + "Wanli Gu", + "Wei Lu", + "Weichang Wu", + "Weiguang Han", + "Weiquan Li", + "Wenbo Shen", + "Wenjing Fang", + "Wenzhi Tang", + "Xiang Shu", + "Xiao Shi", + "Xiaodong Yan", + "Xiaolu Zhang", + "Xiaopei Wan", + "Xiaqing Sun", + "Xin Zhao", + "Xingyu Lu", + "Xinxing Yang", + "Xinyao Tang", + "Xinyu Kong", + "Xinyu Liu", + "Xiong Xu", + "Xuan Sun", + "Xudong Han", + "Xudong Wang", + "Xujie Shen", + "Yalin Zhang", + "Yangyang Hou", + "Yankun Ren", + "Yao Zhao", + "Ye Chen", + "Yeyang Chen", + "Yibo Cao", + "Yifan Zuo", + "Yijie Chen", + "Ying Li", + "Yingjie Song", + "Yingxue Li", + "Yiqi Wang", + "Yixuan Sun", + "Yizhu Xiao", + "Yongfei Xu", + "Yu Liu", + "Yuchen Fang", + "Yue Gao", + "Yue Yu", + "Yue Zhang", + "Yuqi Zhang", + "Yuxiao He", + "Yuxiao Lu", + "Yuxin Tian", + "Yuxuan Li", + "Yuzhuo Fu", + "Zhankai Xu", + "Zhaoxin Huan", + "Zhenduo Zhang", + "Zhengke Gui", + "Zhengyu Huang", + "Zhenjun Ma", + "Zhenxuan Pan", + "Zheping Qu", + "Zhibo Zhu", + "Zhidong Fan", + "Zhigang Huangfu", + "Zhihao Wang", + "Zhiqiang Zhang", + "Zhizhen Liu", + "Zhuyan Zhou", + "Zibin Lin", + "Zihang Zeng", + "Zihao Wang", + "Zilong Wang", + "Ziqi Liu", + "Zitao Xuan", + "Zixuan Cheng", + "Zujie Wen", + "Zuoli Tang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.15079", + "source": "arxiv", + "source_id": "arxiv:2606.15079", + "pdf_url": "https://arxiv.org/pdf/2606.15079", + "primary_query": "tool-use" + }, + { + "id": "2606.14571", + "title": "StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance", + "url": "https://arxiv.org/abs/2606.14571", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Guanming Liu", + "Yuqi Ren", + "Hansu Gu", + "Peng Zhang", + "Weihang Wang", + "Jiahao Liu", + "Ning Gu", + "Tun Lu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.14571", + "source": "arxiv", + "source_id": "arxiv:2606.14571", + "pdf_url": "https://arxiv.org/pdf/2606.14571", + "primary_query": "agent-memory" + }, + { + "id": "2606.13643", + "title": "Recursive Agent Harnesses", + "url": "https://arxiv.org/abs/2606.13643", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Elias Lumer", + "Sahil Sen", + "Kevin Paul", + "Vamse Kumar Subbiah" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2606.13643", + "source": "arxiv", + "source_id": "arxiv:2606.13643", + "pdf_url": "https://arxiv.org/pdf/2606.13643", + "primary_query": "function-calling" + }, + { + "id": "2606.13385", + "title": "Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents", + "url": "https://arxiv.org/abs/2606.13385", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Zihao Wang", + "Yiming Li", + "Yutong Wu", + "Zheyu Liu", + "Kangjie Chen", + "Fok Kar Wai", + "Pin-Yu Chen", + "Vrizlynn L. L. Thing", + "Bo Li", + "Dacheng Tao", + "Tianwei Zhang" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CY", + "cs.HC", + "cs.MM" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.13385", + "source": "arxiv", + "source_id": "arxiv:2606.13385", + "pdf_url": "https://arxiv.org/pdf/2606.13385", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13904", + "title": "SANA: What Matters for QA Agents over Massive Data Lakes?", + "url": "https://arxiv.org/abs/2606.13904", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Austin Senna Wijaya", + "Jiaxiang Liu", + "Haonan Wang", + "Eugene Wu" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.DB" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.13904", + "source": "arxiv", + "source_id": "arxiv:2606.13904", + "pdf_url": "https://arxiv.org/pdf/2606.13904", + "primary_query": "planning-agent" + }, + { + "id": "2606.12341", + "title": "OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents", + "url": "https://arxiv.org/abs/2606.12341", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Jin Xie", + "Songze Li" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.12341", + "source": "arxiv", + "source_id": "arxiv:2606.12341", + "pdf_url": "https://arxiv.org/pdf/2606.12341", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.12195", + "title": "InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning", + "url": "https://arxiv.org/abs/2606.12195", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Ziang Yan", + "Sheng Xia", + "Jiashuo Yu", + "Yue Wu", + "Tianxiang Jiang", + "Songze Li", + "Kanghui Tian", + "Yicheng Xu", + "Yinan He", + "Kai Chen", + "Limin Wang", + "Yu Qiao", + "Yi Wang" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.12195", + "source": "arxiv", + "source_id": "arxiv:2606.12195", + "pdf_url": "https://arxiv.org/pdf/2606.12195", + "primary_query": "tool-use" + }, + { + "id": "2606.11869", + "title": "Agents All the Way Down; A Methodology for Building Custom AI Agents from Substrate to Production", + "url": "https://arxiv.org/abs/2606.11869", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Marc Alier Forment", + "Juanan Pereira", + "Francisco José García-Peñalvo", + "María José Casañ Guerrero" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2606.11869", + "source": "arxiv", + "source_id": "arxiv:2606.11869", + "pdf_url": "https://arxiv.org/pdf/2606.11869", + "primary_query": "function-calling" + }, + { + "id": "2606.17076", + "title": "CMIP-Forge: An Agentic System that Retrieves, Computes, and Self-Reviews Climate Science", + "url": "https://arxiv.org/abs/2606.17076", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Dmitrii Pantiukhin", + "Boris Shapkin", + "Ivan Kuznetsov", + "Thomas Jung", + "Nikolay Koldunov" + ], + "categories": [ + "physics.ao-ph", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.17076", + "source": "arxiv", + "source_id": "arxiv:2606.17076", + "pdf_url": "https://arxiv.org/pdf/2606.17076", + "primary_query": "rag-agent" + }, + { + "id": "2606.11349", + "title": "Knowing When to Ask: Self-Gated Clarification for Hierarchical Language Agents", + "url": "https://arxiv.org/abs/2606.11349", + "published": "2026-06-09", + "updated": "2026-06-12", + "authors": [ + "Aijing Gao", + "Yiming Kang", + "Mengdie Flora Wang", + "Jae Oh Woo" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.11349", + "source": "arxiv", + "source_id": "arxiv:2606.11349", + "pdf_url": "https://arxiv.org/pdf/2606.11349", + "primary_query": "language-agent" + }, + { + "id": "2606.11078", + "title": "A History-Aware Visually Grounded Critic for Computer Use Agents", + "url": "https://arxiv.org/abs/2606.11078", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Jaewoo Lee", + "Zaid Khan", + "Archiki Prasad", + "Justin Chih-Yao Chen", + "Supriyo Chakraborty", + "Kartik Balasubramaniam", + "Sambit Sahu", + "Elias Stengel-Eskin", + "Hyunji Lee", + "Mohit Bansal" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.11078", + "source": "arxiv", + "source_id": "arxiv:2606.11078", + "pdf_url": "https://arxiv.org/pdf/2606.11078", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.10423", + "title": "WebChallenger: A Reliable and Efficient Generalist Web Agent", + "url": "https://arxiv.org/abs/2606.10423", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Jayoo Hwang", + "Xiaowen Zhang", + "Vedant Padwal" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "computer-use", + "embodied-agent", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.10423", + "source": "arxiv", + "source_id": "arxiv:2606.10423", + "pdf_url": "https://arxiv.org/pdf/2606.10423", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.10381", + "title": "Agentic Hybrid RAG for Evidence-Grounded Muon Collider Analysis", + "url": "https://arxiv.org/abs/2606.10381", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Ruobing Jiang", + "Dawei Fu", + "Cheng Jiang", + "Tianyi Yang", + "Zijian Wang", + "Youpeng Wu", + "Yong Ban", + "Yajun Mao", + "Qiang Li" + ], + "categories": [ + "hep-ex", + "cs.AI", + "cs.CL", + "cs.IR", + "physics.ins-det" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.10381", + "source": "arxiv", + "source_id": "arxiv:2606.10381", + "pdf_url": "https://arxiv.org/pdf/2606.10381", + "primary_query": "rag-agent" + }, + { + "id": "2606.09764", + "title": "iOSWorld: A Benchmark for Personally Intelligent Phone Agents", + "url": "https://arxiv.org/abs/2606.09764", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Lawrence Keunho Jang", + "Mareks Woodside", + "Geronimo Carom", + "Andrew Keunwoo Jang", + "Jing Yu Koh", + "Ruslan Salakhutdinov" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "web-gui-agent" + ], + "arxiv_id": "2606.09764", + "source": "arxiv", + "source_id": "arxiv:2606.09764", + "pdf_url": "https://arxiv.org/pdf/2606.09764", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.09399", + "title": "RunAgent SuperBrowser: A Theory of Autonomous Web Navigation Grounded in Human Browsing Behaviour", + "url": "https://arxiv.org/abs/2606.09399", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Radeen Mostafa", + "Sawradip Saha" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.09399", + "source": "arxiv", + "source_id": "arxiv:2606.09399", + "pdf_url": "https://arxiv.org/pdf/2606.09399", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09549", + "title": "SecureClaw: Clawing Back Control of LLM Agents", + "url": "https://arxiv.org/abs/2606.09549", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Yuhan Ma", + "Stefan Schmid" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "planning-agent" + ], + "arxiv_id": "2606.09549", + "source": "arxiv", + "source_id": "arxiv:2606.09549", + "pdf_url": "https://arxiv.org/pdf/2606.09549", + "primary_query": "agent-safety" + }, + { + "id": "2606.09071", + "title": "REFLECT: Intervention-Supported Error Attribution for Silent Failures in LLM Agent Traces", + "url": "https://arxiv.org/abs/2606.09071", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Xiaofeng Lin", + "Yingxu Wang", + "Tung Sum Thomas Kwok", + "Daniel Guo", + "Sahil Arun Nale", + "Charles Fleming", + "Guang Cheng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.09071", + "source": "arxiv", + "source_id": "arxiv:2606.09071", + "pdf_url": "https://arxiv.org/pdf/2606.09071", + "primary_query": "planning-agent" + }, + { + "id": "2606.05463", + "title": "PSEBench: A Controllable and Verifiable Benchmark for Evaluating LLMs in Patient Safety Event Triage", + "url": "https://arxiv.org/abs/2606.05463", + "published": "2026-06-03", + "updated": "2026-06-09", + "authors": [ + "Keqi Han", + "Ryan Young", + "Annabel Strauss", + "Lindsey Hughes", + "Katharine M. Nesbitt", + "Nicole Schueler", + "Che Ngufor", + "Carl Yang", + "Yuan Xue", + "Zhijun Yin" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.05463", + "source": "arxiv", + "source_id": "arxiv:2606.05463", + "pdf_url": "https://arxiv.org/pdf/2606.05463", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.03135", + "title": "Uncertainty-Aware Clarification in LLM Agents with Information Gain", + "url": "https://arxiv.org/abs/2606.03135", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Mengyi Deng", + "Zhiwei Li", + "Xin Li", + "Tingyu Zhu", + "Ying Zhao", + "Zhijiang Guo", + "Wei Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.03135", + "source": "arxiv", + "source_id": "arxiv:2606.03135", + "pdf_url": "https://arxiv.org/pdf/2606.03135", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.04120", + "title": "SaliMory: Orchestrating Cognitive Memory for Conversational Agents", + "url": "https://arxiv.org/abs/2606.04120", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Kai Zhang", + "Xinyuan Zhang", + "Hongda Jiang", + "Shiun-Zu Kuo", + "Hyokun Yun", + "Ejaz Ahmed", + "Shereen Oraby", + "Ziyun Li", + "Sanat Sharma", + "Ann Lee", + "Ahmed A Aly", + "Anuj Kumar", + "Raffay Hamid", + "Xin Luna Dong" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04120", + "source": "arxiv", + "source_id": "arxiv:2606.04120", + "pdf_url": "https://arxiv.org/pdf/2606.04120", + "primary_query": "agent-memory" + }, + { + "id": "2606.04296", + "title": "The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents", + "url": "https://arxiv.org/abs/2606.04296", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Manvendra Modgil" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.04296", + "source": "arxiv", + "source_id": "arxiv:2606.04296", + "pdf_url": "https://arxiv.org/pdf/2606.04296", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.03108", + "title": "EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.03108", + "published": "2026-06-02", + "updated": "2026-06-12", + "authors": [ + "Guhong Chen", + "Yingcheng Shi", + "Yongbin Li", + "Binhua Li", + "Xander Xu", + "Hu Wei", + "Shiwen Ni", + "Min Yang", + "Jieping Ye" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.03108", + "source": "arxiv", + "source_id": "arxiv:2606.03108", + "pdf_url": "https://arxiv.org/pdf/2606.03108", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.02388", + "title": "Policy and World Modeling Co-Training for Language Agents", + "url": "https://arxiv.org/abs/2606.02388", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Ning Lu", + "Baijiong Lin", + "Shengcai Liu", + "Jiahao Wu", + "Haoze Lv", + "Yanbin Wei", + "Lingting Zhu", + "Shengju Qian", + "Xin Wang", + "Ying-Cong Chen", + "Qi Wang", + "Ke Tang" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.02388", + "source": "arxiv", + "source_id": "arxiv:2606.02388", + "pdf_url": "https://arxiv.org/pdf/2606.02388", + "primary_query": "language-agent" + }, + { + "id": "2606.01815", + "title": "CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation", + "url": "https://arxiv.org/abs/2606.01815", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Danqing Wang", + "Akshay Sivaraman", + "Lei Li" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.01815", + "source": "arxiv", + "source_id": "arxiv:2606.01815", + "pdf_url": "https://arxiv.org/pdf/2606.01815", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.02380", + "title": "SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence", + "url": "https://arxiv.org/abs/2606.02380", + "published": "2026-06-01", + "updated": "2026-06-28", + "authors": [ + "Yuyan Bu", + "Haowei Li", + "Qirui Zheng", + "Bowen Dong", + "Kaiyue Yang", + "Jiaming Ji", + "Yingshui Tan", + "Wenxin Li", + "Yaodong Yang", + "Juntao Dai" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.02380", + "source": "arxiv", + "source_id": "arxiv:2606.02380", + "pdf_url": "https://arxiv.org/pdf/2606.02380", + "primary_query": "agent-safety" + }, + { + "id": "2606.00914", + "title": "Adversarial Feeds Steer LLM Agent Decisions Against Their Defaults", + "url": "https://arxiv.org/abs/2606.00914", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Rana Muhammad Usman" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.00914", + "source": "arxiv", + "source_id": "arxiv:2606.00914", + "pdf_url": "https://arxiv.org/pdf/2606.00914", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.00915", + "title": "Autonomous agentic design for photonics", + "url": "https://arxiv.org/abs/2606.00915", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Prashanta Kharel", + "Amin Khavasi", + "Xinzhong Chen", + "Tyler W. Hughes" + ], + "categories": [ + "physics.optics" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.00915", + "source": "arxiv", + "source_id": "arxiv:2606.00915", + "pdf_url": "https://arxiv.org/pdf/2606.00915", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.31278", + "title": "Industrializing Prediction-Powered Inference: The GLIDE Library for Reliable GenAI and Agentic Systems Evaluation", + "url": "https://arxiv.org/abs/2605.31278", + "published": "2026-05-29", + "updated": "2026-06-04", + "authors": [ + "Grégoire Martinon", + "Ibrahim Merad", + "Mohammed Raki" + ], + "categories": [ + "cs.AI", + "cs.LG", + "stat.ME" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.31278", + "source": "arxiv", + "source_id": "arxiv:2605.31278", + "pdf_url": "https://arxiv.org/pdf/2605.31278", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.29676", + "title": "Notation Matters: A Benchmark Study of Token-Optimized Formats in Agentic AI Systems", + "url": "https://arxiv.org/abs/2605.29676", + "published": "2026-05-28", + "updated": "2026-06-17", + "authors": [ + "Lorenz Kutschka", + "Bernhard Geiger" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.29676", + "source": "arxiv", + "source_id": "arxiv:2605.29676", + "pdf_url": "https://arxiv.org/pdf/2605.29676", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.28046", + "title": "MemCog: From Memory-as-Tool to Memory-as-Cognition in Conversational Agents", + "url": "https://arxiv.org/abs/2605.28046", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Zihan Li", + "Xingyu Fan", + "Feifei Li", + "Wenhui Que" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.28046", + "source": "arxiv", + "source_id": "arxiv:2605.28046", + "pdf_url": "https://arxiv.org/pdf/2605.28046", + "primary_query": "agent-memory" + }, + { + "id": "2605.28607", + "title": "Adaptive Multimodal Agents-Based Framework for Automatic Workflow Execution", + "url": "https://arxiv.org/abs/2605.28607", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Susanna Cifani", + "Mario Luca Bernardi", + "Marta Cimitile" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "multi-agent", + "planning", + "rag", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.28607", + "source": "arxiv", + "source_id": "arxiv:2605.28607", + "pdf_url": "https://arxiv.org/pdf/2605.28607", + "primary_query": "rag-agent" + }, + { + "id": "2605.28120", + "title": "LegalGraphRAG: Multi-Agent Graph Retrieval-Augmented Generation for Reliable Legal Reasoning", + "url": "https://arxiv.org/abs/2605.28120", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Zerui Chen", + "Qinggang Zhang", + "Zhishang Xiang", + "Zhimin Wei", + "Linfeng Gao", + "Xiao Huang", + "Zhihong Zhang", + "Jinsong Su" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.28120", + "source": "arxiv", + "source_id": "arxiv:2605.28120", + "pdf_url": "https://arxiv.org/pdf/2605.28120", + "primary_query": "rag-agent" + }, + { + "id": "2605.28787", + "title": "Do Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval", + "url": "https://arxiv.org/abs/2605.28787", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Shiyu Chen", + "Tarfah Alrashed", + "Alon Halevy", + "Natasha Noy" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.28787", + "source": "arxiv", + "source_id": "arxiv:2605.28787", + "pdf_url": "https://arxiv.org/pdf/2605.28787", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.27366", + "title": "MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation", + "url": "https://arxiv.org/abs/2605.27366", + "published": "2026-05-26", + "updated": "2026-07-03", + "authors": [ + "Huawei Lin", + "Peng Li", + "Jie Song", + "Fuxin Jiang", + "Tieying Zhang" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.27366", + "source": "arxiv", + "source_id": "arxiv:2605.27366", + "pdf_url": "https://arxiv.org/pdf/2605.27366", + "primary_query": "agent-memory" + }, + { + "id": "2605.26926", + "title": "From Norms to Indicators (N2I-RAG): An Agentic Retrieval-Augmented Generation Framework for Legal Indicator Computation", + "url": "https://arxiv.org/abs/2605.26926", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Youssef Al Mouatamid", + "Marie Bonnin", + "Jihad Zahir" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.26926", + "source": "arxiv", + "source_id": "arxiv:2605.26926", + "pdf_url": "https://arxiv.org/pdf/2605.26926", + "primary_query": "rag-agent" + }, + { + "id": "2605.27333", + "title": "FinHarness: An Inline Lifecycle Safety Harness for Finance LLM Agents", + "url": "https://arxiv.org/abs/2605.27333", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Haoxuan Jia", + "Yang Liu", + "Bin Chong", + "Yingguang Yang", + "Yancheng Chen", + "Jiayu Liang", + "Qian Li", + "Hanning Lu", + "Kefu Xu", + "Hao Zheng", + "Chongyang Zhang", + "Hao Peng", + "Philip S. Yu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.27333", + "source": "arxiv", + "source_id": "arxiv:2605.27333", + "pdf_url": "https://arxiv.org/pdf/2605.27333", + "primary_query": "planning-agent" + }, + { + "id": "2605.26720", + "title": "Towards Feedback-to-Plan Decisions for Self-Evolving LLM Agents in CUDA Kernel Generation", + "url": "https://arxiv.org/abs/2605.26720", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Yee Hin Chong", + "Jiaming Wu", + "Youhui Zhang", + "Peng Qu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.26720", + "source": "arxiv", + "source_id": "arxiv:2605.26720", + "pdf_url": "https://arxiv.org/pdf/2605.26720", + "primary_query": "planning-agent" + }, + { + "id": "2605.26252", + "title": "Is Agent Memory a Database? Rethinking Data Foundations for Long-Term AI Agent Memory", + "url": "https://arxiv.org/abs/2605.26252", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Abdelghny Orogat", + "Essam Mansour" + ], + "categories": [ + "cs.AI", + "cs.DB" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.26252", + "source": "arxiv", + "source_id": "arxiv:2605.26252", + "pdf_url": "https://arxiv.org/pdf/2605.26252", + "primary_query": "agent-memory" + }, + { + "id": "2605.26305", + "title": "Experiments in Agentic AI for Science", + "url": "https://arxiv.org/abs/2605.26305", + "published": "2026-05-25", + "updated": "2026-05-29", + "authors": [ + "Judy Fox", + "Geoffrey Fox" + ], + "categories": [ + "cs.AI", + "eess.SY", + "hep-ph" + ], + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "rag-agent" + ], + "arxiv_id": "2605.26305", + "source": "arxiv", + "source_id": "arxiv:2605.26305", + "pdf_url": "https://arxiv.org/pdf/2605.26305", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.23636", + "title": "RF Instrument Agent (RFIA): Empowering RF Instruments with Natural Language Understanding, Scheduling and Execution of Complex Tasks", + "url": "https://arxiv.org/abs/2605.23636", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Chunhui Li", + "Wei Fan" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.23636", + "source": "arxiv", + "source_id": "arxiv:2605.23636", + "pdf_url": "https://arxiv.org/pdf/2605.23636", + "primary_query": "language-agent" + }, + { + "id": "2605.22321", + "title": "Benchmarking Autonomous Agents against Temporal, Spatial, and Semantic Evasions", + "url": "https://arxiv.org/abs/2605.22321", + "published": "2026-05-21", + "updated": "2026-05-21", + "authors": [ + "Jianan Ma", + "Xiaohu Du", + "Ruixiao Lin", + "Yaoxiang Bian", + "Jialuo Chen", + "Jingyi Wang", + "Xiaofang Yang", + "Shiwen Cui", + "Changhua Meng", + "Xinhao Deng", + "Zhen Wang" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.22321", + "source": "arxiv", + "source_id": "arxiv:2605.22321", + "pdf_url": "https://arxiv.org/pdf/2605.22321", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.21740", + "title": "SMDD-Bench: Can LLMs Solve Real-World Small Molecule Drug Design Tasks?", + "url": "https://arxiv.org/abs/2605.21740", + "published": "2026-05-20", + "updated": "2026-05-24", + "authors": [ + "Kevin Han", + "Renfei Zhang", + "Kathy Wei", + "Hamed Mahdavi", + "Niloofar Mireshghallah", + "Amir Barati Farimani" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.21740", + "source": "arxiv", + "source_id": "arxiv:2605.21740", + "pdf_url": "https://arxiv.org/pdf/2605.21740", + "primary_query": "planning-agent" + }, + { + "id": "2605.20874", + "title": "Governance by Construction for Generalist Agents", + "url": "https://arxiv.org/abs/2605.20874", + "published": "2026-05-20", + "updated": "2026-05-20", + "authors": [ + "Segev Shlomov", + "Iftach Shoham", + "Alon Oved", + "Ido Levy", + "Sami Marreed", + "Harold Ship", + "Offer Akrabi", + "Sergey Zeltyn", + "Avi Yaeli", + "Nir Mashkif" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-safety", + "computer-use", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.20874", + "source": "arxiv", + "source_id": "arxiv:2605.20874", + "pdf_url": "https://arxiv.org/pdf/2605.20874", + "primary_query": "planning-agent" + }, + { + "id": "2605.20306", + "title": "WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents", + "url": "https://arxiv.org/abs/2605.20306", + "published": "2026-05-19", + "updated": "2026-06-02", + "authors": [ + "Bingnan Liu", + "Chenhang Cui", + "Rui Huang", + "Jiani Luo", + "Zhirong Shen", + "Tinghao Wang", + "Xiande Huang", + "Lingbei Meng", + "Fei Shen", + "An Zhang" + ], + "categories": [ + "cs.CV", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.20306", + "source": "arxiv", + "source_id": "arxiv:2605.20306", + "pdf_url": "https://arxiv.org/pdf/2605.20306", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.18672", + "title": "Position: A Three-Layer Probabilistic Assume-Guarantee Architecture Is Structurally Required for Safe LLM Agent Deployment", + "url": "https://arxiv.org/abs/2605.18672", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "S. Bensalem", + "Y. Dong", + "M. Franzle", + "X. Huang", + "J. Kroger", + "D. Nickovic", + "A. Nouri", + "R. Roy", + "C. Wu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.18672", + "source": "arxiv", + "source_id": "arxiv:2605.18672", + "pdf_url": "https://arxiv.org/pdf/2605.18672", + "primary_query": "agent-safety" + }, + { + "id": "2605.17348", + "title": "Taming \"Zombie'' Agents: A Markov State-Aware Framework for Resilient Multi-Agent Evolution", + "url": "https://arxiv.org/abs/2605.17348", + "published": "2026-05-17", + "updated": "2026-05-17", + "authors": [ + "Taolin Zhang", + "Pukun Zhao", + "Qizhou Chen", + "Jiuheng Wan", + "Chen Chen", + "Xiaofeng He", + "Chengyu Wang", + "Richang Hong" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-safety", + "memory", + "multi-agent", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.17348", + "source": "arxiv", + "source_id": "arxiv:2605.17348", + "pdf_url": "https://arxiv.org/pdf/2605.17348", + "primary_query": "agent-memory" + }, + { + "id": "2605.17453", + "title": "Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback", + "url": "https://arxiv.org/abs/2605.17453", + "published": "2026-05-17", + "updated": "2026-05-17", + "authors": [ + "Lecheng Yan", + "Ruizhe Li", + "Xicheng Han", + "Wenxi Li", + "Binwu Wang", + "Longyue Wang", + "Chenyang Lyu", + "Guanhua Chen" + ], + "categories": [ + "cs.CR", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.17453", + "source": "arxiv", + "source_id": "arxiv:2605.17453", + "pdf_url": "https://arxiv.org/pdf/2605.17453", + "primary_query": "agent-safety" + }, + { + "id": "2605.23986", + "title": "MemForest: An Efficient Agent Memory System with Hierarchical Temporal Indexing", + "url": "https://arxiv.org/abs/2605.23986", + "published": "2026-05-16", + "updated": "2026-05-16", + "authors": [ + "Han Chen", + "Zining Zhang", + "Wenqi Pei", + "Bingsheng He", + "Ming Wu", + "Jason Zeng", + "Michael Heinrich", + "Wei Wu", + "Hongbao Zhang" + ], + "categories": [ + "cs.DB", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.23986", + "source": "arxiv", + "source_id": "arxiv:2605.23986", + "pdf_url": "https://arxiv.org/pdf/2605.23986", + "primary_query": "agent-memory" + }, + { + "id": "2605.28850", + "title": "Representation Signatures and Risk-Feedback Alignment in LLM Trading Agents", + "url": "https://arxiv.org/abs/2605.28850", + "published": "2026-05-16", + "updated": "2026-05-30", + "authors": [ + "Weicheng Xue" + ], + "categories": [ + "cs.LG", + "q-fin.CP" + ], + "topics": [ + "agent-safety", + "coding-agent", + "memory", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.28850", + "source": "arxiv", + "source_id": "arxiv:2605.28850", + "pdf_url": "https://arxiv.org/pdf/2605.28850", + "primary_query": "planning-agent" + }, + { + "id": "2605.14460", + "title": "Exploiting LLM Agent Supply Chains via Payload-less Skills", + "url": "https://arxiv.org/abs/2605.14460", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Xinyu Liu", + "Yukai Zhao", + "Xing Hu", + "Xin Xia" + ], + "categories": [ + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.14460", + "source": "arxiv", + "source_id": "arxiv:2605.14460", + "pdf_url": "https://arxiv.org/pdf/2605.14460", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.14126", + "title": "Reinforcement Learning for Tool-Calling Agents in Fast Healthcare Interoperability Resources (FHIR)", + "url": "https://arxiv.org/abs/2605.14126", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Marius S. Knorr", + "Robert Müller", + "Jan P. Bremer", + "Nils Schweingruber" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.14126", + "source": "arxiv", + "source_id": "arxiv:2605.14126", + "pdf_url": "https://arxiv.org/pdf/2605.14126", + "primary_query": "planning-agent" + }, + { + "id": "2605.11882", + "title": "On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment", + "url": "https://arxiv.org/abs/2605.11882", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Bo Yin", + "Qi Li", + "Xinchao Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.11882", + "source": "arxiv", + "source_id": "arxiv:2605.11882", + "pdf_url": "https://arxiv.org/pdf/2605.11882", + "primary_query": "agent-safety" + }, + { + "id": "2605.11534", + "title": "PRISM: : Planning and Reasoning with Intent in Simulated Embodied Environments", + "url": "https://arxiv.org/abs/2605.11534", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Yunn Kang Lim", + "Pengzhan Sun", + "Ziyi Bai", + "Xun Xu", + "Angela Yao", + "Xulei Yang", + "Shijie Li" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.11534", + "source": "arxiv", + "source_id": "arxiv:2605.11534", + "pdf_url": "https://arxiv.org/pdf/2605.11534", + "primary_query": "planning-agent" + }, + { + "id": "2605.11388", + "title": "Deep Reasoning in General Purpose Agents via Structured Meta-Cognition", + "url": "https://arxiv.org/abs/2605.11388", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Dean Light", + "Michael Theologitis", + "Kshitish Ghate", + "Shuyue Stella Li", + "Benjamin Newman", + "Chirag Shah", + "Aylin Caliskan", + "Pang Wei Koh", + "Dan Suciu", + "Yulia Tsvetkov" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.11388", + "source": "arxiv", + "source_id": "arxiv:2605.11388", + "pdf_url": "https://arxiv.org/pdf/2605.11388", + "primary_query": "planning-agent" + }, + { + "id": "2605.11039", + "title": "The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck", + "url": "https://arxiv.org/abs/2605.11039", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Linfeng Fan", + "Ziwei Li", + "Yuan Tian", + "Yichen Wang", + "Rongsheng Li", + "Xiong Wang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.11039", + "source": "arxiv", + "source_id": "arxiv:2605.11039", + "pdf_url": "https://arxiv.org/pdf/2605.11039", + "primary_query": "agent-safety" + }, + { + "id": "2605.10763", + "title": "MATRA: Modeling the Attack Surface of Agentic AI Systems -- OpenClaw Case Study", + "url": "https://arxiv.org/abs/2605.10763", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Tim Van hamme", + "Thomas Vissers", + "Javier Carnerero-Cano", + "Mario Fritz", + "Emil C. Lupu", + "Lieven Desmet", + "Dinil Mon Divakaran" + ], + "categories": [ + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.10763", + "source": "arxiv", + "source_id": "arxiv:2605.10763", + "pdf_url": "https://arxiv.org/pdf/2605.10763", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.10365", + "title": "Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values", + "url": "https://arxiv.org/abs/2605.10365", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Haonan Dong", + "Qiguan Feng", + "Kehan Jiang", + "Haoran Ye", + "Xin Zhang", + "Guojie Song" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.10365", + "source": "arxiv", + "source_id": "arxiv:2605.10365", + "pdf_url": "https://arxiv.org/pdf/2605.10365", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.09168", + "title": "CIVeX: Causal Intervention Verification for Language Agents", + "url": "https://arxiv.org/abs/2605.09168", + "published": "2026-05-09", + "updated": "2026-05-09", + "authors": [ + "Fabio Rovai" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.09168", + "source": "arxiv", + "source_id": "arxiv:2605.09168", + "pdf_url": "https://arxiv.org/pdf/2605.09168", + "primary_query": "language-agent" + }, + { + "id": "2605.08763", + "title": "When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions", + "url": "https://arxiv.org/abs/2605.08763", + "published": "2026-05-09", + "updated": "2026-05-09", + "authors": [ + "Minfeng Qi", + "Tianqing Zhu", + "Zijie Xu", + "Congcong Zhu", + "Qin Wang", + "Wanlei Zhou" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.08763", + "source": "arxiv", + "source_id": "arxiv:2605.08763", + "pdf_url": "https://arxiv.org/pdf/2605.08763", + "primary_query": "planning-agent" + }, + { + "id": "2605.06078", + "title": "Milestone-Guided Policy Learning for Long-Horizon Language Agents", + "url": "https://arxiv.org/abs/2605.06078", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Zixuan Wang", + "Yuchen Yan", + "Hongxing Li", + "Teng Pan", + "Dingming Li", + "Ruiqing Zhang", + "Weiming Lu", + "Jun Xiao", + "Yueting Zhuang", + "Yongliang Shen" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.06078", + "source": "arxiv", + "source_id": "arxiv:2605.06078", + "pdf_url": "https://arxiv.org/pdf/2605.06078", + "primary_query": "language-agent" + }, + { + "id": "2605.03505", + "title": "LATS-RCA: Language Agent Tree Search for Root Cause Analysis in Microservices", + "url": "https://arxiv.org/abs/2605.03505", + "published": "2026-05-05", + "updated": "2026-06-12", + "authors": [ + "Alexander Naakka", + "Yuqing Wang", + "Mika V Mäntylä" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.03505", + "source": "arxiv", + "source_id": "arxiv:2605.03505", + "pdf_url": "https://arxiv.org/pdf/2605.03505", + "primary_query": "language-agent" + }, + { + "id": "2605.04107", + "title": "TSCG: Deterministic Tool-Schema Compilation for Agentic LLM Deployments", + "url": "https://arxiv.org/abs/2605.04107", + "published": "2026-05-04", + "updated": "2026-05-04", + "authors": [ + "Furkan Sakizli" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.04107", + "source": "arxiv", + "source_id": "arxiv:2605.04107", + "pdf_url": "https://arxiv.org/pdf/2605.04107", + "primary_query": "function-calling" + }, + { + "id": "2605.05242", + "title": "Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction", + "url": "https://arxiv.org/abs/2605.05242", + "published": "2026-05-03", + "updated": "2026-05-03", + "authors": [ + "Zhuofeng Li", + "Haoxiang Zhang", + "Cong Wei", + "Pan Lu", + "Ping Nie", + "Yi Lu", + "Yuyang Bai", + "Shangbin Feng", + "Hangxiao Zhu", + "Ming Zhong", + "Yuyu Zhang", + "Jianwen Xie", + "Yejin Choi", + "James Zou", + "Jiawei Han", + "Wenhu Chen", + "Jimmy Lin", + "Dongfu Jiang", + "Yu Zhang" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.05242", + "source": "arxiv", + "source_id": "arxiv:2605.05242", + "pdf_url": "https://arxiv.org/pdf/2605.05242", + "primary_query": "language-agent" + }, + { + "id": "2605.00081", + "title": "Alignment Contracts for Agentic Security Systems", + "url": "https://arxiv.org/abs/2605.00081", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Isaac David", + "Marco Guarnieri", + "Arthur Gervais" + ], + "categories": [ + "cs.CR", + "cs.LO" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.00081", + "source": "arxiv", + "source_id": "arxiv:2605.00081", + "pdf_url": "https://arxiv.org/pdf/2605.00081", + "primary_query": "agent-safety" + }, + { + "id": "2604.27699", + "title": "Bridging Values and Behavior: A Hierarchical Framework for Proactive Embodied Agents", + "url": "https://arxiv.org/abs/2604.27699", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Chunhui Zhang", + "Yuxuan Wang", + "Aoyang Qin", + "Yi-Long Lu", + "Kunlun Wu", + "Yizhou Wang", + "Wei Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27699", + "source": "arxiv", + "source_id": "arxiv:2604.27699", + "pdf_url": "https://arxiv.org/pdf/2604.27699", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.26274", + "title": "Enforcing Benign Trajectories: A Behavioral Firewall for Structured-Workflow AI Agents", + "url": "https://arxiv.org/abs/2604.26274", + "published": "2026-04-29", + "updated": "2026-04-29", + "authors": [ + "Hung Dang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.26274", + "source": "arxiv", + "source_id": "arxiv:2604.26274", + "pdf_url": "https://arxiv.org/pdf/2604.26274", + "primary_query": "agent-safety" + }, + { + "id": "2604.24826", + "title": "A Comparative Evaluation of AI Agent Security Guardrails", + "url": "https://arxiv.org/abs/2604.24826", + "published": "2026-04-27", + "updated": "2026-04-27", + "authors": [ + "Qi Li", + "Jiu Li", + "Pingtao Wei", + "Jianjun Xu", + "Xueyi Wei", + "Jiwei Shi", + "Xuan Zhang", + "Yanhui Yang", + "Xiaodong Hui", + "Peng Xu", + "Lingquan Zhou" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.24826", + "source": "arxiv", + "source_id": "arxiv:2604.24826", + "pdf_url": "https://arxiv.org/pdf/2604.24826", + "primary_query": "agent-safety" + }, + { + "id": "2606.13686", + "title": "Benchmarking Web Agent Safety under E-commerce Deceptive Interfaces", + "url": "https://arxiv.org/abs/2606.13686", + "published": "2026-04-26", + "updated": "2026-04-26", + "authors": [ + "Zijing Shi", + "Meng Fang", + "Ling Chen" + ], + "categories": [ + "cs.CL", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.13686", + "source": "arxiv", + "source_id": "arxiv:2606.13686", + "pdf_url": "https://arxiv.org/pdf/2606.13686", + "primary_query": "agent-safety" + }, + { + "id": "2604.19821", + "title": "JTPRO: A Joint Tool-Prompt Reflective Optimization Framework for Language Agents", + "url": "https://arxiv.org/abs/2604.19821", + "published": "2026-04-20", + "updated": "2026-04-20", + "authors": [ + "Sandip Ghoshal", + "Anshul Mittal", + "Jyotika Singh", + "Miguel Ballesteros", + "Weiyi Sun", + "Fang Tu", + "Shailender Singh", + "Yassine Benajiba", + "Fahad Shah", + "Sujeeth Bharadwaj", + "Sujith Ravi", + "Dan Roth" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.19821", + "source": "arxiv", + "source_id": "arxiv:2604.19821", + "pdf_url": "https://arxiv.org/pdf/2604.19821", + "primary_query": "language-agent" + }, + { + "id": "2604.18718", + "title": "Towards Optimal Agentic Architectures for Offensive Security Tasks", + "url": "https://arxiv.org/abs/2604.18718", + "published": "2026-04-20", + "updated": "2026-04-20", + "authors": [ + "Isaac David", + "Arthur Gervais" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.18718", + "source": "arxiv", + "source_id": "arxiv:2604.18718", + "pdf_url": "https://arxiv.org/pdf/2604.18718", + "primary_query": "agent-safety" + }, + { + "id": "2604.12986", + "title": "Parallax: Why AI Agents That Think Must Never Act", + "url": "https://arxiv.org/abs/2604.12986", + "published": "2026-04-14", + "updated": "2026-04-14", + "authors": [ + "Joel Fokou" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.12986", + "source": "arxiv", + "source_id": "arxiv:2604.12986", + "pdf_url": "https://arxiv.org/pdf/2604.12986", + "primary_query": "agent-safety" + }, + { + "id": "2604.06972", + "title": "Differentiable Environment-Trajectory Co-Optimization for Safe Multi-Agent Navigation", + "url": "https://arxiv.org/abs/2604.06972", + "published": "2026-04-08", + "updated": "2026-04-08", + "authors": [ + "Zhan Gao", + "Gabriele Fadini", + "Stelian Coros", + "Amanda Prorok" + ], + "categories": [ + "cs.RO", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "multi-agent", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.06972", + "source": "arxiv", + "source_id": "arxiv:2604.06972", + "pdf_url": "https://arxiv.org/pdf/2604.06972", + "primary_query": "agent-safety" + }, + { + "id": "2604.04426", + "title": "ShieldNet: Network-Level Guardrails against Emerging Supply-Chain Injections in Agentic Systems", + "url": "https://arxiv.org/abs/2604.04426", + "published": "2026-04-06", + "updated": "2026-04-06", + "authors": [ + "Zhuowen Yuan", + "Zhaorun Chen", + "Zhen Xiang", + "Nathaniel D. Bastian", + "Seyyed Hadi Hashemi", + "Chaowei Xiao", + "Wenbo Guo", + "Bo Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.04426", + "source": "arxiv", + "source_id": "arxiv:2604.04426", + "pdf_url": "https://arxiv.org/pdf/2604.04426", + "primary_query": "agent-safety" + }, + { + "id": "2604.03098", + "title": "Co-Evolution of Policy and Internal Reward for Language Agents", + "url": "https://arxiv.org/abs/2604.03098", + "published": "2026-04-03", + "updated": "2026-04-03", + "authors": [ + "Xinyu Wang", + "Hanwei Wu", + "Jingwei Song", + "Shuyuan Zhang", + "Jiayi Zhang", + "Fanqi Kong", + "Tung Sum Thomas Kwok", + "Xiao-Wen Chang", + "Yuyu Luo", + "Chenglin Wu", + "Bang Liu" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.03098", + "source": "arxiv", + "source_id": "arxiv:2604.03098", + "pdf_url": "https://arxiv.org/pdf/2604.03098", + "primary_query": "language-agent" + }, + { + "id": "2603.15309", + "title": "CCTU: A Benchmark for Tool Use under Complex Constraints", + "url": "https://arxiv.org/abs/2603.15309", + "published": "2026-03-16", + "updated": "2026-03-16", + "authors": [ + "Junjie Ye", + "Guoqiang Zhang", + "Wenjie Fu", + "Tao Gui", + "Qi Zhang", + "Xuanjing Huang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.15309", + "source": "arxiv", + "source_id": "arxiv:2603.15309", + "pdf_url": "https://arxiv.org/pdf/2603.15309", + "primary_query": "function-calling" + }, + { + "id": "2603.11890", + "title": "QUARE: Quality-Aware Requirements Analysis through Multi-Agent Dialectical Negotiation", + "url": "https://arxiv.org/abs/2603.11890", + "published": "2026-03-12", + "updated": "2026-06-05", + "authors": [ + "Haowei Cheng", + "Milhan Kim", + "Foutse Khomh", + "Teeradaj Racharak", + "Nobukazu Yoshioka", + "Naoyasu Ubayashi", + "Hironori Washizaki" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.11890", + "source": "arxiv", + "source_id": "arxiv:2603.11890", + "pdf_url": "https://arxiv.org/pdf/2603.11890", + "primary_query": "agent-safety" + }, + { + "id": "2603.07557", + "title": "AgentRaft: Automated Detection of Data Over-Exposure in LLM Agents", + "url": "https://arxiv.org/abs/2603.07557", + "published": "2026-03-08", + "updated": "2026-03-08", + "authors": [ + "Yixi Lin", + "Jiangrong Wu", + "Yuhong Nan", + "Xueqiang Wang", + "Xinyuan Zhang", + "Zibin Zheng" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.07557", + "source": "arxiv", + "source_id": "arxiv:2603.07557", + "pdf_url": "https://arxiv.org/pdf/2603.07557", + "primary_query": "function-calling" + }, + { + "id": "2603.05578", + "title": "Tool-Genesis: A Task-Driven Tool Creation Benchmark for Self-Evolving Language Agent", + "url": "https://arxiv.org/abs/2603.05578", + "published": "2026-03-05", + "updated": "2026-03-05", + "authors": [ + "Bowei Xia", + "Mengkang Hu", + "Shijian Wang", + "Jiarui Jin", + "Wenxiang Jiao", + "Yuan Lu", + "Kexin Li", + "Ping Luo" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.05578", + "source": "arxiv", + "source_id": "arxiv:2603.05578", + "pdf_url": "https://arxiv.org/pdf/2603.05578", + "primary_query": "language-agent" + }, + { + "id": "2603.01712", + "title": "FT-Dojo: Towards Autonomous LLM Fine-Tuning with Language Agents", + "url": "https://arxiv.org/abs/2603.01712", + "published": "2026-03-02", + "updated": "2026-05-20", + "authors": [ + "Qizheng Li", + "Yifei Zhang", + "Xiao Yang", + "Xu Yang", + "Zhuo Wang", + "Weiqing Liu", + "Jiang Bian" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.01712", + "source": "arxiv", + "source_id": "arxiv:2603.01712", + "pdf_url": "https://arxiv.org/pdf/2603.01712", + "primary_query": "language-agent" + }, + { + "id": "2602.13379", + "title": "Unsafer in Many Turns: Benchmarking and Defending Multi-Turn Safety Risks in Tool-Using Agents", + "url": "https://arxiv.org/abs/2602.13379", + "published": "2026-02-13", + "updated": "2026-06-10", + "authors": [ + "Xu Li", + "Simon Yu", + "Minzhou Pan", + "Yiyou Sun", + "Bo Li", + "Dawn Song", + "Xue Lin", + "Weiyan Shi" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.13379", + "source": "arxiv", + "source_id": "arxiv:2602.13379", + "pdf_url": "https://arxiv.org/pdf/2602.13379", + "primary_query": "agent-safety" + }, + { + "id": "2602.11749", + "title": "AIR: Improving Agent Safety through Incident Response", + "url": "https://arxiv.org/abs/2602.11749", + "published": "2026-02-12", + "updated": "2026-06-20", + "authors": [ + "Zibo Xiao", + "Jun Sun", + "Junjie Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.11749", + "source": "arxiv", + "source_id": "arxiv:2602.11749", + "pdf_url": "https://arxiv.org/pdf/2602.11749", + "primary_query": "agent-safety" + }, + { + "id": "2602.18456", + "title": "Beyond single-channel agentic benchmarking", + "url": "https://arxiv.org/abs/2602.18456", + "published": "2026-02-05", + "updated": "2026-02-05", + "authors": [ + "Nelu D. Radpour" + ], + "categories": [ + "cs.CY", + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.18456", + "source": "arxiv", + "source_id": "arxiv:2602.18456", + "pdf_url": "https://arxiv.org/pdf/2602.18456", + "primary_query": "agent-safety" + }, + { + "id": "2601.14652", + "title": "MAS-Orchestra: Understanding and Improving Multi-Agent Reasoning Through Holistic Orchestration and Controlled Benchmarks", + "url": "https://arxiv.org/abs/2601.14652", + "published": "2026-01-21", + "updated": "2026-05-21", + "authors": [ + "Zixuan Ke", + "Yifei Ming", + "Austin Xu", + "Ryan Chin", + "Xuan-Phi Nguyen", + "Prathyusha Jwalapuram", + "Jiayu Wang", + "Semih Yavuz", + "Caiming Xiong", + "Shafiq Joty" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.14652", + "source": "arxiv", + "source_id": "arxiv:2601.14652", + "pdf_url": "https://arxiv.org/pdf/2601.14652", + "primary_query": "function-calling" + }, + { + "id": "2512.23647", + "title": "Nested Browser-Use Learning for Agentic Information Seeking", + "url": "https://arxiv.org/abs/2512.23647", + "published": "2025-12-29", + "updated": "2025-12-29", + "authors": [ + "Baixuan Li", + "Jialong Wu", + "Wenbiao Yin", + "Kuan Li", + "Zhongwang Zhang", + "Huifeng Yin", + "Zhengwei Tao", + "Liwen Zhang", + "Pengjun Xie", + "Jingren Zhou", + "Yong Jiang" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.23647", + "source": "arxiv", + "source_id": "arxiv:2512.23647", + "pdf_url": "https://arxiv.org/pdf/2512.23647", + "primary_query": "function-calling" + }, + { + "id": "2512.23611", + "title": "Close the Loop: Synthesizing Infinite Tool-Use Data via Multi-Agent Role-Playing", + "url": "https://arxiv.org/abs/2512.23611", + "published": "2025-12-29", + "updated": "2025-12-29", + "authors": [ + "Yuwen Li", + "Wei Zhang", + "Zelong Huang", + "Mason Yang", + "Jiajun Wu", + "Shawn Guo", + "Huahao Hu", + "Lingyi Sun", + "Jian Yang", + "Mingjie Tang", + "Byran Dai" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.23611", + "source": "arxiv", + "source_id": "arxiv:2512.23611", + "pdf_url": "https://arxiv.org/pdf/2512.23611", + "primary_query": "function-calling" + }, + { + "id": "2512.02605", + "title": "IACT: A Self-Organizing Recursive Model for General AI Agents: A Technical White Paper on the Architecture Behind kragent.ai", + "url": "https://arxiv.org/abs/2512.02605", + "published": "2025-12-02", + "updated": "2025-12-02", + "authors": [ + "Pengju Lu" + ], + "categories": [ + "cs.AI", + "cs.MA", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.02605", + "source": "arxiv", + "source_id": "arxiv:2512.02605", + "pdf_url": "https://arxiv.org/pdf/2512.02605", + "primary_query": "function-calling" + }, + { + "id": "2510.14548", + "title": "LLM Agents Beyond Utility: An Open-Ended Perspective", + "url": "https://arxiv.org/abs/2510.14548", + "published": "2025-10-16", + "updated": "2025-10-16", + "authors": [ + "Asen Nachkov", + "Xi Wang", + "Luc Van Gool" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.14548", + "source": "arxiv", + "source_id": "arxiv:2510.14548", + "pdf_url": "https://arxiv.org/pdf/2510.14548", + "primary_query": "function-calling" + }, + { + "id": "2509.26553", + "title": "Towards Reliable Benchmarking: A Contamination Free, Controllable Evaluation Framework for Multi-step LLM Function Calling", + "url": "https://arxiv.org/abs/2509.26553", + "published": "2025-09-30", + "updated": "2026-02-06", + "authors": [ + "Seiji Maekawa", + "Jackson Hassell", + "Pouya Pezeshkpour", + "Tom Mitchell", + "Estevam Hruschka" + ], + "categories": [ + "cs.CL", + "cs.PL" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.26553", + "source": "arxiv", + "source_id": "arxiv:2509.26553", + "pdf_url": "https://arxiv.org/pdf/2509.26553", + "primary_query": "function-calling" + }, + { + "id": "2509.14477", + "title": "Ticket-Bench: A Kickoff for Multilingual and Regionalized Agent Evaluation", + "url": "https://arxiv.org/abs/2509.14477", + "published": "2025-09-17", + "updated": "2025-09-17", + "authors": [ + "Thales Sales Almeida", + "João Guilherme Alves Santos", + "Thiago Laitz", + "Giovana Kerche Bonás" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.14477", + "source": "arxiv", + "source_id": "arxiv:2509.14477", + "pdf_url": "https://arxiv.org/pdf/2509.14477", + "primary_query": "function-calling" + }, + { + "id": "2509.02444", + "title": "AppCopilot: Toward General, Accurate, Long-Horizon, and Efficient Mobile Agent", + "url": "https://arxiv.org/abs/2509.02444", + "published": "2025-09-02", + "updated": "2025-10-17", + "authors": [ + "Jingru Fan", + "Yufan Dang", + "Jingyao Wu", + "Huatao Li", + "Runde Yang", + "Xiyuan Yang", + "Yuheng Wang", + "Chen Qian" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CV", + "cs.HC" + ], + "topics": [ + "computer-use", + "memory", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 15, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.02444", + "source": "arxiv", + "source_id": "arxiv:2509.02444", + "pdf_url": "https://arxiv.org/pdf/2509.02444", + "primary_query": "function-calling" + }, + { + "id": "2607.06140", + "title": "CurateEvo: Data-Curation Evolving for Agentic Post-Training", + "url": "https://arxiv.org/abs/2607.06140", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Dingzirui Wang", + "Xuanliang Zhang", + "Keyan Xu", + "Qingfu Zhu", + "Wanxiang Che" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "memory", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.06140", + "source": "arxiv", + "source_id": "arxiv:2607.06140", + "pdf_url": "https://arxiv.org/pdf/2607.06140", + "primary_query": "llm-agent" + }, + { + "id": "2607.06452", + "title": "From Voting to Agent Collaboration: Answer-Type-Aware LLM Pipelines for BioASQ 14b", + "url": "https://arxiv.org/abs/2607.06452", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Taeyun Roh", + "Eunha Lee", + "Wonjune Jang", + "Sohyun Chung", + "Junha Jung", + "Jaewoo Kang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.06452", + "source": "arxiv", + "source_id": "arxiv:2607.06452", + "pdf_url": "https://arxiv.org/pdf/2607.06452", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.05772", + "title": "Detecting Vulnerability-Inducing Commits via Multi-Stage Reasoning with LLM-Based Agents", + "url": "https://arxiv.org/abs/2607.05772", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Liyou Chen", + "Hailong Sun", + "Xiang Gao", + "Yue Pan" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.05772", + "source": "arxiv", + "source_id": "arxiv:2607.05772", + "pdf_url": "https://arxiv.org/pdf/2607.05772", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.05378", + "title": "CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents", + "url": "https://arxiv.org/abs/2607.05378", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Yujiang Li", + "Zhenyu Hou", + "Yi Jing", + "Jie Tang", + "Yuxiao Dong" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "coding-agent", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent", + "llm-agent" + ], + "arxiv_id": "2607.05378", + "source": "arxiv", + "source_id": "arxiv:2607.05378", + "pdf_url": "https://arxiv.org/pdf/2607.05378", + "primary_query": "coding-agent" + }, + { + "id": "2607.05132", + "title": "When Agents Lie: Premeditation, Persistence, and Exploitation in Repeated Games", + "url": "https://arxiv.org/abs/2607.05132", + "published": "2026-07-06", + "updated": "2026-07-07", + "authors": [ + "Jerick Shi", + "Terry Jingcheng Zhang", + "Bernhard Schölkopf", + "Vincent Conitzer", + "Zhijing Jin" + ], + "categories": [ + "cs.CY", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.05132", + "source": "arxiv", + "source_id": "arxiv:2607.05132", + "pdf_url": "https://arxiv.org/pdf/2607.05132", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.04963", + "title": "STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training", + "url": "https://arxiv.org/abs/2607.04963", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Qiuyi Qi", + "Tian Liang", + "Mutian Bao", + "Jinjian Zhang", + "Dongnan Liu", + "Wei Zhou", + "Linjian Mo", + "Ming Kong", + "Jie Liu", + "Feng Zhang", + "Qiang Zhu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "planning", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.04963", + "source": "arxiv", + "source_id": "arxiv:2607.04963", + "pdf_url": "https://arxiv.org/pdf/2607.04963", + "primary_query": "llm-agent" + }, + { + "id": "2607.05677", + "title": "From Conversation to Contribution: Characterizing Coding Agent in Open-Source Software", + "url": "https://arxiv.org/abs/2607.05677", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Zihan Fang", + "Yueke Zhang", + "Ningzhi Tang", + "Collin McMillan", + "Toby Jia-Jun Li", + "Yu Huang" + ], + "categories": [ + "cs.SE", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.05677", + "source": "arxiv", + "source_id": "arxiv:2607.05677", + "pdf_url": "https://arxiv.org/pdf/2607.05677", + "primary_query": "coding-agent" + }, + { + "id": "2607.05188", + "title": "Latent Programming Horizons in Coding Agents", + "url": "https://arxiv.org/abs/2607.05188", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "André Silva", + "Han Tu", + "Martin Monperrus" + ], + "categories": [ + "cs.LG", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.05188", + "source": "arxiv", + "source_id": "arxiv:2607.05188", + "pdf_url": "https://arxiv.org/pdf/2607.05188", + "primary_query": "coding-agent" + }, + { + "id": "2607.04623", + "title": "Can LLMs Really Recover Microservice Failures? A Recovery-Aware Evaluation of Diagnosis-to-Action Reasoning", + "url": "https://arxiv.org/abs/2607.04623", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Jiaxing Qi", + "Zhongzhi Luan", + "Hongyu Zhang", + "Shaohan Huang", + "Carol Fung", + "Yongxin Tong", + "Hailong Yang", + "Depei Qian" + ], + "categories": [ + "cs.SE", + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.04623", + "source": "arxiv", + "source_id": "arxiv:2607.04623", + "pdf_url": "https://arxiv.org/pdf/2607.04623", + "primary_query": "rag-agent" + }, + { + "id": "2607.04470", + "title": "Regime-Conditional Stabilisation of LLM-Augmented Cooperative Multi-Agent Reinforcement Learning", + "url": "https://arxiv.org/abs/2607.04470", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Faid Keddouri", + "Sohaib Houhou", + "Aissa Boulmerka", + "Nadir Farhi" + ], + "categories": [ + "cs.LG", + "cs.AI", + "math.OC" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.04470", + "source": "arxiv", + "source_id": "arxiv:2607.04470", + "pdf_url": "https://arxiv.org/pdf/2607.04470", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.03853", + "title": "CogRad: A Cognitively-Inspired Multi-Agent Framework for Radiology Report Generation", + "url": "https://arxiv.org/abs/2607.03853", + "published": "2026-07-04", + "updated": "2026-07-04", + "authors": [ + "Saif Ur Rehman Khan", + "Hasaan Maqsood", + "Sebastian Vollmer", + "Andreas Dengel", + "Muhammad Nabeel Asim" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.03853", + "source": "arxiv", + "source_id": "arxiv:2607.03853", + "pdf_url": "https://arxiv.org/pdf/2607.03853", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.02879", + "title": "MedCalc-Pro: Solving Complex Medical Calculations with LLM Agents", + "url": "https://arxiv.org/abs/2607.02879", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Siran Zhao", + "Ruihui Hou", + "Ziyue Huai", + "Chennuo Zhang", + "Tong Ruan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.02879", + "source": "arxiv", + "source_id": "arxiv:2607.02879", + "pdf_url": "https://arxiv.org/pdf/2607.02879", + "primary_query": "llm-agent" + }, + { + "id": "2607.03423", + "title": "Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies", + "url": "https://arxiv.org/abs/2607.03423", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Chris Schneider", + "Kriti Faujdar", + "Philipp Schoenegger", + "Ben Bariach" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2607.03423", + "source": "arxiv", + "source_id": "arxiv:2607.03423", + "pdf_url": "https://arxiv.org/pdf/2607.03423", + "primary_query": "ai-agent" + }, + { + "id": "2607.03162", + "title": "APeB: Benchmarking Personalization Ability of Large Language Model Agents", + "url": "https://arxiv.org/abs/2607.03162", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Garry Yang", + "Zizhe Chen", + "Xinru Chen", + "Yongqiang Chen", + "Jianxiang Wang", + "Deyu Zou", + "Linyi Ding", + "Jialiang Wu", + "Yunzhong He", + "Yu Gong", + "James Cheng", + "Huaixiao Tou" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.03162", + "source": "arxiv", + "source_id": "arxiv:2607.03162", + "pdf_url": "https://arxiv.org/pdf/2607.03162", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02882", + "title": "Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference", + "url": "https://arxiv.org/abs/2607.02882", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Xuyan Ma", + "Yawen Wang", + "Junjie Wang", + "Xiaofei Xie", + "Boyu Wu", + "Mingyang Li", + "Dandan Wang", + "Qing Wang" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02882", + "source": "arxiv", + "source_id": "arxiv:2607.02882", + "pdf_url": "https://arxiv.org/pdf/2607.02882", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03628", + "title": "Swarm-Driven Multi-Agent Reasoning for Smart City Security", + "url": "https://arxiv.org/abs/2607.03628", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Saeid Jamshidi", + "Kawser Wazed Nafi", + "Carol Fung", + "Foutse Khomh" + ], + "categories": [ + "cs.CR", + "cs.MA" + ], + "topics": [ + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.03628", + "source": "arxiv", + "source_id": "arxiv:2607.03628", + "pdf_url": "https://arxiv.org/pdf/2607.03628", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.01600", + "title": "BOUNDARY_SYNC: Measuring Communication-Induced Representational Coupling in Multi-Agent LLM Systems", + "url": "https://arxiv.org/abs/2607.01600", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Zewen Liu" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "multi-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "multi-agent-llm" + ], + "arxiv_id": "2607.01600", + "source": "arxiv", + "source_id": "arxiv:2607.01600", + "pdf_url": "https://arxiv.org/pdf/2607.01600", + "primary_query": "llm-agent" + }, + { + "id": "2607.02210", + "title": "Criticality-Based Guard Rail Validation for AI Agent Decisions in Autonomous Telecom Networks", + "url": "https://arxiv.org/abs/2607.02210", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Ravi Kant Sharma" + ], + "categories": [ + "cs.AI", + "cs.NI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.02210", + "source": "arxiv", + "source_id": "arxiv:2607.02210", + "pdf_url": "https://arxiv.org/pdf/2607.02210", + "primary_query": "ai-agent" + }, + { + "id": "2607.01661", + "title": "Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry", + "url": "https://arxiv.org/abs/2607.01661", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Yuante Li", + "Yicheng Tao", + "Kate Zhang", + "Taozhi Wang", + "Gefei Gu", + "Yaxin Zhou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.01661", + "source": "arxiv", + "source_id": "arxiv:2607.01661", + "pdf_url": "https://arxiv.org/pdf/2607.01661", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00692", + "title": "Self-GC: Self-Governing Context for Long-Horizon LLM Agents", + "url": "https://arxiv.org/abs/2607.00692", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Xubin Hao", + "Hongjin Meng", + "Xin Yin", + "Jiawei Zhu", + "Chenpeng Cao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "planning", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.00692", + "source": "arxiv", + "source_id": "arxiv:2607.00692", + "pdf_url": "https://arxiv.org/pdf/2607.00692", + "primary_query": "llm-agent" + }, + { + "id": "2607.00297", + "title": "EPC: A Standardized Protocol for Measuring Evaluator Preference Dynamics in LLM Agent Systems", + "url": "https://arxiv.org/abs/2607.00297", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Zewen Liu" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.00297", + "source": "arxiv", + "source_id": "arxiv:2607.00297", + "pdf_url": "https://arxiv.org/pdf/2607.00297", + "primary_query": "llm-agent" + }, + { + "id": "2607.01523", + "title": "Multi-Head Recurrent Memory Agents", + "url": "https://arxiv.org/abs/2607.01523", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Jiatong Li", + "Samuel Yeh", + "Sharon Li" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2607.01523", + "source": "arxiv", + "source_id": "arxiv:2607.01523", + "pdf_url": "https://arxiv.org/pdf/2607.01523", + "primary_query": "agent-memory" + }, + { + "id": "2607.01211", + "title": "Are Performance-Optimization Benchmarks Reliably Measuring Coding Agents?", + "url": "https://arxiv.org/abs/2607.01211", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Zhi Chen", + "Zhensu Sun", + "Yuling Shi", + "David Lo", + "Lingxiao Jiang" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.01211", + "source": "arxiv", + "source_id": "arxiv:2607.01211", + "pdf_url": "https://arxiv.org/pdf/2607.01211", + "primary_query": "coding-agent" + }, + { + "id": "2607.00918", + "title": "From Personas to Plot: Character-Grounded Multi-Agent Story Generation for Long-Form Narratives", + "url": "https://arxiv.org/abs/2607.00918", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Aayush Aluru", + "Chloe Ho", + "Muhammad Hammouri", + "Kerry Luo", + "Myra Malik", + "Ryan Lagasse", + "Arjun Bahuguna", + "Vasu Sharma" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00918", + "source": "arxiv", + "source_id": "arxiv:2607.00918", + "pdf_url": "https://arxiv.org/pdf/2607.00918", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00604", + "title": "Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives", + "url": "https://arxiv.org/abs/2607.00604", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Xianchao Xiu", + "Chong Shen", + "Yanjiao Zhu", + "Wanquan Liu" + ], + "categories": [ + "math.OC" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.00604", + "source": "arxiv", + "source_id": "arxiv:2607.00604", + "pdf_url": "https://arxiv.org/pdf/2607.00604", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.00972", + "title": "Bayesian Uncertainty Propagation for Agentic RAG Pipelines: A Proof-of-Concept Study on Multi-Hop Question Answering", + "url": "https://arxiv.org/abs/2607.00972", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Louis Donaldson", + "Connor Walker", + "Koorosh Aslansefat", + "Yiannis Papadopoulos" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.00972", + "source": "arxiv", + "source_id": "arxiv:2607.00972", + "pdf_url": "https://arxiv.org/pdf/2607.00972", + "primary_query": "rag-agent" + }, + { + "id": "2607.00422", + "title": "KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems", + "url": "https://arxiv.org/abs/2607.00422", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Chanwoo Choi", + "Euntae Kim", + "Kyuho Lee", + "Youngsam Chun", + "Jinhee Jeong", + "Eunmi Kim", + "Myunggyo Oh", + "Junseo Jang", + "Buru Chang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2607.00422", + "source": "arxiv", + "source_id": "arxiv:2607.00422", + "pdf_url": "https://arxiv.org/pdf/2607.00422", + "primary_query": "rag-agent" + }, + { + "id": "2606.31635", + "title": "A Tutorial on Autonomous Fault-Tolerant Control Using Knowledge-Grounded LLM Agents", + "url": "https://arxiv.org/abs/2606.31635", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Javal Vyas", + "Milapji Singh Gill", + "Artan Markaj", + "Felix Gehlhoff", + "Mehmet Mercangöz" + ], + "categories": [ + "eess.SY", + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-safety", + "planning", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.31635", + "source": "arxiv", + "source_id": "arxiv:2606.31635", + "pdf_url": "https://arxiv.org/pdf/2606.31635", + "primary_query": "llm-agent" + }, + { + "id": "2606.31227", + "title": "Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming", + "url": "https://arxiv.org/abs/2606.31227", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Yong Yang", + "Xing Zheng", + "Huiyu Wu", + "Huangsheng Cheng", + "Xiaorong Shi", + "Jing Guo", + "Bo Yang", + "Yi Zhou", + "Xiangfan Wu", + "Zonghao Ying" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "ai-agent" + ], + "arxiv_id": "2606.31227", + "source": "arxiv", + "source_id": "arxiv:2606.31227", + "pdf_url": "https://arxiv.org/pdf/2606.31227", + "primary_query": "agent-safety" + }, + { + "id": "2607.00255", + "title": "SLM, LLM or Agentic AI? Toward Intelligent UAV-Enabled WPT Systems in Low-Altitude Economy Networks", + "url": "https://arxiv.org/abs/2607.00255", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Feibo Jiang", + "Li Dong", + "Lei Mao", + "Kezhi Wang", + "Xianbin Wang", + "Abbas Jamalipour" + ], + "categories": [ + "cs.IT" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.00255", + "source": "arxiv", + "source_id": "arxiv:2607.00255", + "pdf_url": "https://arxiv.org/pdf/2607.00255", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31648", + "title": "Think in English, Answer in Korean: Efficient Adaptation of Multilingual Tool-Using Agents", + "url": "https://arxiv.org/abs/2606.31648", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Utsav Garg", + "Sungjin Hong", + "Jason Jung", + "Justin Lee", + "Shaan Desai", + "Joon Hee Kim", + "Anirudh Shrinivason", + "Edmond Wen", + "Susie Park" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "memory", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "function-calling", + "tool-use" + ], + "arxiv_id": "2606.31648", + "source": "arxiv", + "source_id": "arxiv:2606.31648", + "pdf_url": "https://arxiv.org/pdf/2606.31648", + "primary_query": "agentic-ai" + }, + { + "id": "2607.02577", + "title": "Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation", + "url": "https://arxiv.org/abs/2607.02577", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Vishvesh Bhat", + "Jay Vaghasiya", + "Muhammad Ahmed Mohsin", + "Asad Aali" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.02577", + "source": "arxiv", + "source_id": "arxiv:2607.02577", + "pdf_url": "https://arxiv.org/pdf/2607.02577", + "primary_query": "tool-use" + }, + { + "id": "2606.31314", + "title": "A Novel Method for Differential-Algebraic Dynamic Model Discovery in Power Systems: An LLM-Based Multi-Agent Collaborative Framework", + "url": "https://arxiv.org/abs/2606.31314", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Xinming Wang", + "Fan Tang", + "Yingli Wei", + "Yakun He", + "Zhe Liu", + "Ping Jiang", + "Haoyu Wu", + "Zihan Guo", + "Chao Shen" + ], + "categories": [ + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31314", + "source": "arxiv", + "source_id": "arxiv:2606.31314", + "pdf_url": "https://arxiv.org/pdf/2606.31314", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.30840", + "title": "Contrastive Reflection for Iterative Prompt Optimization", + "url": "https://arxiv.org/abs/2606.30840", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Derek Koh", + "Jinghui Mo", + "Benjamin H. Le", + "Jiening Zhan", + "Baofen Zheng", + "Kevin Bevis", + "Nathaniel C. Owen", + "Lauren Elizabeth Charney", + "Wenqiong Liu", + "Jingwei Wu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent" + ], + "arxiv_id": "2606.30840", + "source": "arxiv", + "source_id": "arxiv:2606.30840", + "pdf_url": "https://arxiv.org/pdf/2606.30840", + "primary_query": "ai-agent" + }, + { + "id": "2606.30454", + "title": "Collective cooperation without individual fidelity in LLM agents", + "url": "https://arxiv.org/abs/2606.30454", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Henrique Ferraz de Arruda", + "Carlos Gracia Lázaro", + "Alberto Aleta", + "Yamir Moreno" + ], + "categories": [ + "physics.soc-ph", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.30454", + "source": "arxiv", + "source_id": "arxiv:2606.30454", + "pdf_url": "https://arxiv.org/pdf/2606.30454", + "primary_query": "llm-agent" + }, + { + "id": "2606.30005", + "title": "LLM Agents Are Latent Context Managers: Eliciting Self-Managed Context via a Proprioceptive Dashboard", + "url": "https://arxiv.org/abs/2606.30005", + "published": "2026-06-29", + "updated": "2026-07-05", + "authors": [ + "Binyan Xu", + "Haitao Li", + "Kehuan Zhang" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "memory", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.30005", + "source": "arxiv", + "source_id": "arxiv:2606.30005", + "pdf_url": "https://arxiv.org/pdf/2606.30005", + "primary_query": "llm-agent" + }, + { + "id": "2606.29762", + "title": "Do Recommendation Algorithms Work When Users Are LLM Agents? A Case Study on Moltbook", + "url": "https://arxiv.org/abs/2606.29762", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Daming Li", + "Simeng Han", + "Jialu Zhang" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "llm-agent" + ], + "arxiv_id": "2606.29762", + "source": "arxiv", + "source_id": "arxiv:2606.29762", + "pdf_url": "https://arxiv.org/pdf/2606.29762", + "primary_query": "ai-agent" + }, + { + "id": "2606.30266", + "title": "Towards Continual Motion-Language Agents: LoRA Variants for Incremental Motion Understanding and Generation", + "url": "https://arxiv.org/abs/2606.30266", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Bertram Taetz", + "Hugo Albuquerque Cosme da Silva", + "Gabriele Bleser-Taetz" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "language-agent" + ], + "arxiv_id": "2606.30266", + "source": "arxiv", + "source_id": "arxiv:2606.30266", + "pdf_url": "https://arxiv.org/pdf/2606.30266", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.30185", + "title": "Dynamo: Dynamic Skill-Tool Evolution for Vision-Language Agents", + "url": "https://arxiv.org/abs/2606.30185", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Yutao Sun", + "Yanting Miao", + "Hao-Xuan Ma", + "Mengyu Zhou", + "Mingshuai Chen", + "Tiancheng Zhao", + "Dexin Wang", + "Lei Lv", + "Li Xu", + "Xiaoxi Jiang", + "Guanjun Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.30185", + "source": "arxiv", + "source_id": "arxiv:2606.30185", + "pdf_url": "https://arxiv.org/pdf/2606.30185", + "primary_query": "language-agent" + }, + { + "id": "2606.30755", + "title": "Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens", + "url": "https://arxiv.org/abs/2606.30755", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Peizhi Niu", + "Wenjie Qu", + "Shangding Gu", + "Tianneng Shi", + "Yuankai Li", + "Ahmad Tawaha", + "Hend Alzahrani", + "Vincent Siu", + "Boyi Li", + "Chenguang Wang", + "Jiaheng Zhang", + "Basel Alomair", + "Ming Jin", + "Muhao Chen", + "Chi Wang", + "Costas Spanos", + "Dawn Song" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "ai-agent" + ], + "arxiv_id": "2606.30755", + "source": "arxiv", + "source_id": "arxiv:2606.30755", + "pdf_url": "https://arxiv.org/pdf/2606.30755", + "primary_query": "agent-safety" + }, + { + "id": "2606.29894", + "title": "SABER-Math: Automated Benchmark for Information Retrieval Evaluation in Mathematics", + "url": "https://arxiv.org/abs/2606.29894", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Nikolay Georgiev", + "Maria Drencheva", + "Kseniia Ibragimova", + "Ivo Petrov", + "Dimitar I. Dimitrov", + "Martin Vechev" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.29894", + "source": "arxiv", + "source_id": "arxiv:2606.29894", + "pdf_url": "https://arxiv.org/pdf/2606.29894", + "primary_query": "agentic-ai" + }, + { + "id": "2606.29961", + "title": "DuoMem: Towards Capable On-Device Memory Agents via Dual-Space Distillation", + "url": "https://arxiv.org/abs/2606.29961", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Peyman Hosseini", + "Ondrej Bohdal", + "Ahmed Alajrami", + "Andrea Maracani", + "Ignacio Castro", + "Matthew Purver", + "Mete Ozay", + "Savas Ozkan", + "Taha Ceritli" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.29961", + "source": "arxiv", + "source_id": "arxiv:2606.29961", + "pdf_url": "https://arxiv.org/pdf/2606.29961", + "primary_query": "agent-memory" + }, + { + "id": "2606.30119", + "title": "On the Internet, Nobody Knows You're an LLM Bot: Unmasking Web Agents with Multi-Layer Fingerprinting", + "url": "https://arxiv.org/abs/2606.30119", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Iliana Fayolle", + "Sihem Bouhenniche", + "Samuel Pélissier", + "Pierre Laperdrix", + "Clémentine Maurice", + "Walter Rudametkin" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.30119", + "source": "arxiv", + "source_id": "arxiv:2606.30119", + "pdf_url": "https://arxiv.org/pdf/2606.30119", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.30602", + "title": "MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.30602", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Kunyang Li", + "Kyle Domico", + "Jonathan Gregory", + "Patrick McDaniel" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.30602", + "source": "arxiv", + "source_id": "arxiv:2606.30602", + "pdf_url": "https://arxiv.org/pdf/2606.30602", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29932", + "title": "SAGA: Scene-Aware, Goal-Evolving Agents for Long-Horizon CivRealm Strategy Planning", + "url": "https://arxiv.org/abs/2606.29932", + "published": "2026-06-29", + "updated": "2026-07-02", + "authors": [ + "Tianyu Jin", + "Shuo Chen", + "Yida Wang", + "Liuyu Xiang", + "Yingzhuo Liu", + "Zhiyao Jiang", + "Yexin Li", + "Zhaofeng He" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.29932", + "source": "arxiv", + "source_id": "arxiv:2606.29932", + "pdf_url": "https://arxiv.org/pdf/2606.29932", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29746", + "title": "DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification", + "url": "https://arxiv.org/abs/2606.29746", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Maolin Liu", + "Fanyu Xu", + "Ruoqing Xu", + "Jiahang Zhang", + "Hao Wang", + "Rui Wang" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.29746", + "source": "arxiv", + "source_id": "arxiv:2606.29746", + "pdf_url": "https://arxiv.org/pdf/2606.29746", + "primary_query": "rag-agent" + }, + { + "id": "2606.29225", + "title": "PolicyGuard: A Dialogue-Grounded Sub-Agent Verifier for Policy Adherence in LLM Agents", + "url": "https://arxiv.org/abs/2606.29225", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Seongjae Kang", + "Taehyung Yu", + "Sung Ju Hwang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29225", + "source": "arxiv", + "source_id": "arxiv:2606.29225", + "pdf_url": "https://arxiv.org/pdf/2606.29225", + "primary_query": "llm-agent" + }, + { + "id": "2606.29142", + "title": "Agent Security Meets Regulatory Reality -- A Practitioner Systematization of Autonomous-Agent Threats and Controls in Regulated Financial Systems", + "url": "https://arxiv.org/abs/2606.29142", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Krishna Mohan", + "Guda Nagavenkata Srinivasa" + ], + "categories": [ + "cs.CY", + "cs.SE" + ], + "topics": [ + "agent-safety", + "computer-use", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "autonomous-agent-llm", + "rag-agent" + ], + "arxiv_id": "2606.29142", + "source": "arxiv", + "source_id": "arxiv:2606.29142", + "pdf_url": "https://arxiv.org/pdf/2606.29142", + "primary_query": "agent-safety" + }, + { + "id": "2606.28733", + "title": "Agentic Abstention: Do Agents Know When to Stop Instead of Act?", + "url": "https://arxiv.org/abs/2606.28733", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Han Luo", + "Bingbing Wen", + "Lucy Lu Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.28733", + "source": "arxiv", + "source_id": "arxiv:2606.28733", + "pdf_url": "https://arxiv.org/pdf/2606.28733", + "primary_query": "llm-agent" + }, + { + "id": "2606.28679", + "title": "Capability Gates Are Not Authorization: Confused-Deputy Failures in LLM Agent Frameworks", + "url": "https://arxiv.org/abs/2606.28679", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "David Mellafe Zuvic" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "tool-use" + ], + "arxiv_id": "2606.28679", + "source": "arxiv", + "source_id": "arxiv:2606.28679", + "pdf_url": "https://arxiv.org/pdf/2606.28679", + "primary_query": "llm-agent" + }, + { + "id": "2606.29026", + "title": "Preventing Error Propagation in Multi-Agent AI through Runtime Monitoring", + "url": "https://arxiv.org/abs/2606.29026", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Shahnewaz Karim Sakib", + "Anindya Bijoy Das" + ], + "categories": [ + "cs.AI", + "cs.ET" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.29026", + "source": "arxiv", + "source_id": "arxiv:2606.29026", + "pdf_url": "https://arxiv.org/pdf/2606.29026", + "primary_query": "agentic-ai" + }, + { + "id": "2606.28739", + "title": "Agent Safety Is Action Alignment", + "url": "https://arxiv.org/abs/2606.28739", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Shawn Li", + "Yue Zhao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.28739", + "source": "arxiv", + "source_id": "arxiv:2606.28739", + "pdf_url": "https://arxiv.org/pdf/2606.28739", + "primary_query": "agent-safety" + }, + { + "id": "2606.27632", + "title": "Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety", + "url": "https://arxiv.org/abs/2606.27632", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Ting Ma", + "Xiufeng Huang", + "Benlei Cui", + "Xiaowen Xu", + "Shikai Qiu", + "Ruijie Jian", + "Hongxing Li", + "Guanghui Wang", + "Longtao Huang", + "Haiwen Hong", + "Haolei Xu", + "Wenjing Jiang", + "Ziwen Xu", + "Zhaoyu Fan", + "Shaoxuan He", + "Chuxi Xiao", + "Yujian Li", + "Xinyue Chen", + "Chunyang Chai", + "Wenxuan Liu", + "Ziheng Wang", + "Dongjie Zhang", + "Yangfan Zhou", + "Libin Dong", + "Yupeng Cao", + "Xiaoqian Xia", + "Jing Wang", + "Zhe Jiang", + "Zhenan Ye", + "Guang Yang", + "Bin Liu", + "Wei Peng", + "Ziqiang Zhu", + "Meihui Lian", + "Kaiwen Lv Kacuila", + "Haidong Ding", + "Bingyu Zhu", + "Yan Wang", + "Hai Zhao", + "Xuan Jin", + "Wei Zhao", + "Pengfei Sun", + "Wei Wang", + "Huiming Zhang", + "Bin Li", + "Hui Xue" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.27632", + "source": "arxiv", + "source_id": "arxiv:2606.27632", + "pdf_url": "https://arxiv.org/pdf/2606.27632", + "primary_query": "tool-use" + }, + { + "id": "2606.26806", + "title": "Memory Depth, Not Memory Access: Selective Parametric Consolidation for Long-Running Language Agents", + "url": "https://arxiv.org/abs/2606.26806", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Haoliang Han" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.26806", + "source": "arxiv", + "source_id": "arxiv:2606.26806", + "pdf_url": "https://arxiv.org/pdf/2606.26806", + "primary_query": "language-agent" + }, + { + "id": "2606.27154", + "title": "OpenRCA 2.0: From Outcome Labels to Causal Process Supervision", + "url": "https://arxiv.org/abs/2606.27154", + "published": "2026-06-25", + "updated": "2026-06-30", + "authors": [ + "Aoyang Fang", + "Yifan Yang", + "Jin'ao Shang", + "Qisheng Lu", + "Junjielung Xu", + "Rui Wang", + "Songhan Zhang", + "Yuzhong Zhang", + "Boxi Yu", + "Pinjia He" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.27154", + "source": "arxiv", + "source_id": "arxiv:2606.27154", + "pdf_url": "https://arxiv.org/pdf/2606.27154", + "primary_query": "tool-use" + }, + { + "id": "2606.27492", + "title": "QueenBee Planner: Skill-Evolving Communication Topologies for Token-Efficient LLM Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.27492", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Congjia Tian", + "Yuhang Yao", + "Jiaming Cui" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.27492", + "source": "arxiv", + "source_id": "arxiv:2606.27492", + "pdf_url": "https://arxiv.org/pdf/2606.27492", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.26758", + "title": "EGG: An Expert-Guided Agent Framework for Kernel Generation", + "url": "https://arxiv.org/abs/2606.26758", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Yaochen Han", + "Ke Fan", + "Hongxu Jiang", + "Wanqi Xu", + "Weiyu Xie", + "Runhua Zhang", + "Chenhui Zhu", + "Yixiang Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "memory", + "multi-agent", + "rag", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.26758", + "source": "arxiv", + "source_id": "arxiv:2606.26758", + "pdf_url": "https://arxiv.org/pdf/2606.26758", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.26205", + "title": "Knowledge-augmented Agentic AI for Mental Health Medication Information Seeking", + "url": "https://arxiv.org/abs/2606.26205", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Huizi Yu", + "Jian Liu", + "Wenkong Wang", + "Lingyao Li", + "Jiayan Zhou", + "Zhaoqian Xue", + "Xiang Li", + "Xinxin Lin", + "Zhiying Liang", + "Zhuoru Wu", + "Siyuan Ma", + "Xin Ma", + "Lizhou Fan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "multi-agent-llm" + ], + "arxiv_id": "2606.26205", + "source": "arxiv", + "source_id": "arxiv:2606.26205", + "pdf_url": "https://arxiv.org/pdf/2606.26205", + "primary_query": "agentic-ai" + }, + { + "id": "2606.25899", + "title": "Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs", + "url": "https://arxiv.org/abs/2606.25899", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Adeeb Zaman", + "Erik Nordby", + "Fred Heiding" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.25899", + "source": "arxiv", + "source_id": "arxiv:2606.25899", + "pdf_url": "https://arxiv.org/pdf/2606.25899", + "primary_query": "agentic-ai" + }, + { + "id": "2606.25484", + "title": "From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities", + "url": "https://arxiv.org/abs/2606.25484", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Meng Jin", + "Melanie Handrich", + "Simone Martinenz", + "Nicholas Hoeser", + "Ziyue Li" + ], + "categories": [ + "cs.CY", + "econ.GN", + "stat.AP" + ], + "topics": [ + "coding-agent", + "computer-use", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.25484", + "source": "arxiv", + "source_id": "arxiv:2606.25484", + "pdf_url": "https://arxiv.org/pdf/2606.25484", + "primary_query": "agentic-ai" + }, + { + "id": "2606.26300", + "title": "The Verification Horizon: No Silver Bullet for Coding Agent Rewards", + "url": "https://arxiv.org/abs/2606.26300", + "published": "2026-06-24", + "updated": "2026-06-29", + "authors": [ + "Binghai Wang", + "Chenlong Zhang", + "Dayiheng Liu", + "Jiajun Zhang", + "Jiawei Chen", + "Mingze Li", + "Mouxiang Chen", + "Rongyao Fang", + "Siyuan Zhang", + "Xuwu Wang", + "Yuheng Jing", + "Zeyao Ma", + "Zeyu Cui" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.26300", + "source": "arxiv", + "source_id": "arxiv:2606.26300", + "pdf_url": "https://arxiv.org/pdf/2606.26300", + "primary_query": "coding-agent" + }, + { + "id": "2606.25705", + "title": "GUI agent: Guided Exploration of User-Sensitive Screens", + "url": "https://arxiv.org/abs/2606.25705", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Aradhana Nayak", + "Mussadiq Nazeer", + "Wang Peng", + "Feng Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.25705", + "source": "arxiv", + "source_id": "arxiv:2606.25705", + "pdf_url": "https://arxiv.org/pdf/2606.25705", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.25656", + "title": "Is GraphRAG Needed? From Basic RAG to Graph-/Agentic Solutions with Context Optimization", + "url": "https://arxiv.org/abs/2606.25656", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Long Chen", + "Ryan Razkenari", + "Yuxuan Zhou", + "Yuan Tian", + "Rahul Ghosh", + "Venkatesh Pappakrishnan", + "Disha Ahuja", + "Vidya Sagar Ravipati" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.25656", + "source": "arxiv", + "source_id": "arxiv:2606.25656", + "pdf_url": "https://arxiv.org/pdf/2606.25656", + "primary_query": "rag-agent" + }, + { + "id": "2606.25189", + "title": "ActPlane: Programmable OS-Level Policy Enforcement for Agent Harnesses", + "url": "https://arxiv.org/abs/2606.25189", + "published": "2026-06-23", + "updated": "2026-06-30", + "authors": [ + "Yusheng Zheng", + "Tianyuan Wu", + "Quanzhi Fu", + "Tong Yu", + "Wenan Mao", + "Tao Ma", + "Dan Williams", + "Wei Wang", + "Andi Quinn" + ], + "categories": [ + "cs.OS" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.25189", + "source": "arxiv", + "source_id": "arxiv:2606.25189", + "pdf_url": "https://arxiv.org/pdf/2606.25189", + "primary_query": "ai-agent" + }, + { + "id": "2606.24402", + "title": "Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents", + "url": "https://arxiv.org/abs/2606.24402", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Juho Park", + "Hyunmin Choi", + "Kevin Nam" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "rag-agent" + ], + "arxiv_id": "2606.24402", + "source": "arxiv", + "source_id": "arxiv:2606.24402", + "pdf_url": "https://arxiv.org/pdf/2606.24402", + "primary_query": "ai-agent" + }, + { + "id": "2606.24235", + "title": "SP-Mind: An Autonomous Reasoning Agent for Spatial Proteomics Analysis", + "url": "https://arxiv.org/abs/2606.24235", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yucheng Yuan", + "Yuanfeng Ji", + "Zhongxiao Li", + "Ruijiang Li" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.24235", + "source": "arxiv", + "source_id": "arxiv:2606.24235", + "pdf_url": "https://arxiv.org/pdf/2606.24235", + "primary_query": "ai-agent" + }, + { + "id": "2606.25206", + "title": "RAVEN: Long-Horizon Reasoning & Navigation with a Visuo-Spatio-Temporal Memory", + "url": "https://arxiv.org/abs/2606.25206", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yixun Hu", + "Zhicheng Zheng", + "Lihan Zha", + "Chunwei Xing", + "Rajdeep Singh", + "Omar Hossain", + "Antonio Loquercio", + "Dhruv Shah" + ], + "categories": [ + "cs.RO", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.25206", + "source": "arxiv", + "source_id": "arxiv:2606.25206", + "pdf_url": "https://arxiv.org/pdf/2606.25206", + "primary_query": "agent-memory" + }, + { + "id": "2606.24515", + "title": "Reinforcement Learning for Computer-Use Agents with Autonomous Evaluation", + "url": "https://arxiv.org/abs/2606.24515", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Marta Sumyk", + "Oleksandr Kosovan" + ], + "categories": [ + "cs.AI", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.24515", + "source": "arxiv", + "source_id": "arxiv:2606.24515", + "pdf_url": "https://arxiv.org/pdf/2606.24515", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.24694", + "title": "SupplyNet: Supporting Visual Exploratory Learning in Supply Chain via Contextual Multi-Agent Simulation", + "url": "https://arxiv.org/abs/2606.24694", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Yanjia Li", + "Kelcy Kexin Han", + "Tianrui Hu", + "Yi-Fan Cao", + "Huamin Qu", + "Sicheng Song" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "multi-agent", + "rag", + "reasoning", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.24694", + "source": "arxiv", + "source_id": "arxiv:2606.24694", + "pdf_url": "https://arxiv.org/pdf/2606.24694", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.24976", + "title": "Diagnosing and Mitigating Compounding Failures in Agentic Persuasion via Taxonomic Strategy Retrieval", + "url": "https://arxiv.org/abs/2606.24976", + "published": "2026-06-23", + "updated": "2026-07-04", + "authors": [ + "Sana Ayromlou", + "Purvi Sehgal", + "Pradyumna Narayana" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.24976", + "source": "arxiv", + "source_id": "arxiv:2606.24976", + "pdf_url": "https://arxiv.org/pdf/2606.24976", + "primary_query": "rag-agent" + }, + { + "id": "2606.22737", + "title": "GroundEval: A Deterministic Replacement for LLM-as-Judge in Stateful Agent Evaluation", + "url": "https://arxiv.org/abs/2606.22737", + "published": "2026-06-22", + "updated": "2026-07-02", + "authors": [ + "Jeffrey Flynt" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.22737", + "source": "arxiv", + "source_id": "arxiv:2606.22737", + "pdf_url": "https://arxiv.org/pdf/2606.22737", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.23283", + "title": "Towards Root Memories: Benchmarking and Enhancing Implicit Logical Memory Retrieval for Personalized LLMs", + "url": "https://arxiv.org/abs/2606.23283", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Hongxun Ding", + "Xiang Yu", + "Chengbing Wang", + "Jianfei Xiao", + "Keqin Bao", + "Wenjie Wang", + "Xiangnan He" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.23283", + "source": "arxiv", + "source_id": "arxiv:2606.23283", + "pdf_url": "https://arxiv.org/pdf/2606.23283", + "primary_query": "agent-memory" + }, + { + "id": "2606.23195", + "title": "Memory Contagion: Cross-Temporal Propagation of Evaluator Bias via Agent Memory", + "url": "https://arxiv.org/abs/2606.23195", + "published": "2026-06-22", + "updated": "2026-06-24", + "authors": [ + "Zewen Liu" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.23195", + "source": "arxiv", + "source_id": "arxiv:2606.23195", + "pdf_url": "https://arxiv.org/pdf/2606.23195", + "primary_query": "agent-memory" + }, + { + "id": "2606.22864", + "title": "When AUC 0.998 Is Not Enough: A Candidate Evaluation Protocol for Hidden-State Probes of Indirect Prompt Injection in Multimodal Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.22864", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Yanhang Li", + "Zhichao Fan", + "Zexin Zhuang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.22864", + "source": "arxiv", + "source_id": "arxiv:2606.22864", + "pdf_url": "https://arxiv.org/pdf/2606.22864", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.22495", + "title": "Grounded Scaling: Why Agentic AI Needs Deterministic Environments", + "url": "https://arxiv.org/abs/2606.22495", + "published": "2026-06-21", + "updated": "2026-06-21", + "authors": [ + "Liang Ding", + "Xintong Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "embodied-agent", + "multi-agent", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.22495", + "source": "arxiv", + "source_id": "arxiv:2606.22495", + "pdf_url": "https://arxiv.org/pdf/2606.22495", + "primary_query": "agentic-ai" + }, + { + "id": "2606.22484", + "title": "Governed AI-Assisted Engineering: Graduated Human Oversight for Agentic Code Generation in Regulated Domains", + "url": "https://arxiv.org/abs/2606.22484", + "published": "2026-06-21", + "updated": "2026-07-04", + "authors": [ + "Richard Kang" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.22484", + "source": "arxiv", + "source_id": "arxiv:2606.22484", + "pdf_url": "https://arxiv.org/pdf/2606.22484", + "primary_query": "agentic-ai" + }, + { + "id": "2606.22030", + "title": "Nous: A Predictive World Model for Long-Term Agent Memory", + "url": "https://arxiv.org/abs/2606.22030", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Pranav Singh" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.IR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.22030", + "source": "arxiv", + "source_id": "arxiv:2606.22030", + "pdf_url": "https://arxiv.org/pdf/2606.22030", + "primary_query": "agent-memory" + }, + { + "id": "2606.22151", + "title": "Novelty-Aware Agentic Retrieval: Comparing Research Contributions Through Structured Multi-Step Reasoning", + "url": "https://arxiv.org/abs/2606.22151", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Shou-Tzu Han" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.22151", + "source": "arxiv", + "source_id": "arxiv:2606.22151", + "pdf_url": "https://arxiv.org/pdf/2606.22151", + "primary_query": "rag-agent" + }, + { + "id": "2606.21842", + "title": "Agent-Assisted Side-Channel Attacks on Non-Prefix KV Cache in RAG", + "url": "https://arxiv.org/abs/2606.21842", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "He Sun", + "Shinan Liu", + "Siyuan Ma", + "Junhao Li", + "Mingjun Xiao", + "Wenhao Jiang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.21842", + "source": "arxiv", + "source_id": "arxiv:2606.21842", + "pdf_url": "https://arxiv.org/pdf/2606.21842", + "primary_query": "rag-agent" + }, + { + "id": "2606.21409", + "title": "Don't Blindly Trust It: How Unreliable Feedback Breaks Tool-Using LLM Agents", + "url": "https://arxiv.org/abs/2606.21409", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Chubin Zhang", + "Zhenglin Wan", + "Xingrui Yu", + "Pengfei Zhou", + "Wangbo Zhao", + "Jingxuan Wu", + "Yaxin Zhou", + "Ivor Tsang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.21409", + "source": "arxiv", + "source_id": "arxiv:2606.21409", + "pdf_url": "https://arxiv.org/pdf/2606.21409", + "primary_query": "tool-use" + }, + { + "id": "2606.21553", + "title": "Dissecting Agentic RAG: A Component Ablation for Multi-Hop QA with a Local 7B Model", + "url": "https://arxiv.org/abs/2606.21553", + "published": "2026-06-19", + "updated": "2026-06-19", + "authors": [ + "Sheroz Shaikh" + ], + "categories": [ + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.21553", + "source": "arxiv", + "source_id": "arxiv:2606.21553", + "pdf_url": "https://arxiv.org/pdf/2606.21553", + "primary_query": "rag-agent" + }, + { + "id": "2606.20470", + "title": "Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems", + "url": "https://arxiv.org/abs/2606.20470", + "published": "2026-06-18", + "updated": "2026-06-26", + "authors": [ + "Reza Soosahabi", + "Vivek Namsani" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.20470", + "source": "arxiv", + "source_id": "arxiv:2606.20470", + "pdf_url": "https://arxiv.org/pdf/2606.20470", + "primary_query": "agentic-ai" + }, + { + "id": "2606.19812", + "title": "Human-on-the-Loop Orchestration for AI-Assisted Legal Discovery", + "url": "https://arxiv.org/abs/2606.19812", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Anushree Sinha", + "Srivaths Ranganathan", + "Abhishek Dharmaratnakar", + "Debanshu Das" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "workflow-agent", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "planning-agent" + ], + "arxiv_id": "2606.19812", + "source": "arxiv", + "source_id": "arxiv:2606.19812", + "pdf_url": "https://arxiv.org/pdf/2606.19812", + "primary_query": "agentic-ai" + }, + { + "id": "2606.20515", + "title": "S-Agent: Spatial Tool-Use Elicits Reasoning for Spatial Intelligence", + "url": "https://arxiv.org/abs/2606.20515", + "published": "2026-06-18", + "updated": "2026-06-28", + "authors": [ + "Yalun Dai", + "Hao Li", + "Shulin Tian", + "Runmao Yao", + "Yuhao Dong", + "Fangzhou Hong", + "Zhaoxi Chen", + "Fangfu Liu", + "Baoliang Tian", + "Dingwen Zhang", + "Tao Wang", + "Kim-Hui Yap", + "Ziwei Liu" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "tool-use" + ], + "arxiv_id": "2606.20515", + "source": "arxiv", + "source_id": "arxiv:2606.20515", + "pdf_url": "https://arxiv.org/pdf/2606.20515", + "primary_query": "agent-memory" + }, + { + "id": "2606.20023", + "title": "When Lower Privileges Suffice: Investigating Over-Privileged Tool Selection in LLM Agents", + "url": "https://arxiv.org/abs/2606.20023", + "published": "2026-06-18", + "updated": "2026-07-07", + "authors": [ + "Kaiyue Yang", + "Yuyan Bu", + "Jingwei Yi", + "Yuchi Wang", + "Biyu Zhou", + "Juntao Dai", + "Songlin Hu", + "Yaodong Yang" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.20023", + "source": "arxiv", + "source_id": "arxiv:2606.20023", + "pdf_url": "https://arxiv.org/pdf/2606.20023", + "primary_query": "tool-use" + }, + { + "id": "2606.20785", + "title": "Fara-1.5: Scalable Learning Environments for Computer Use Agents", + "url": "https://arxiv.org/abs/2606.20785", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Ahmed Awadallah", + "Sahil Gupta", + "Yash Lara", + "Yadong Lu", + "Hussein Mozannar", + "Akshay Nambi", + "Zach Nussbaum", + "Yash Pandya", + "Aravind Rajeswaran", + "Corby Rosset", + "Alexey Taymanov", + "Luiz do Valle", + "Vibhav Vineet", + "Spencer Whitehead", + "Andrew Zhao" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20785", + "source": "arxiv", + "source_id": "arxiv:2606.20785", + "pdf_url": "https://arxiv.org/pdf/2606.20785", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.19930", + "title": "MobileForge: Annotation-Free Adaptation for Mobile GUI Agents with Hierarchical Feedback-Guided Policy Optimization", + "url": "https://arxiv.org/abs/2606.19930", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Guangyi Liu", + "Pengxiang Zhao", + "Gao Wu", + "Yiwen Yin", + "Mading Li", + "Liang Liu", + "Congxiao Liu", + "Zhang Qi", + "Mengyan Wang", + "Liang Guo", + "Yong Liu" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.19930", + "source": "arxiv", + "source_id": "arxiv:2606.19930", + "pdf_url": "https://arxiv.org/pdf/2606.19930", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.18671", + "title": "HANSEL: Extracting Breadcrumbs from Web Agent Trajectories for Interactive Verification", + "url": "https://arxiv.org/abs/2606.18671", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Yujin Zhang", + "Daye Nam" + ], + "categories": [ + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "planning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "web-gui-agent" + ], + "arxiv_id": "2606.18671", + "source": "arxiv", + "source_id": "arxiv:2606.18671", + "pdf_url": "https://arxiv.org/pdf/2606.18671", + "primary_query": "ai-agent" + }, + { + "id": "2606.19063", + "title": "PYPILINE: Malicious PyPI Package Detection via Suspicious API Knowledge and Agent Workflow", + "url": "https://arxiv.org/abs/2606.19063", + "published": "2026-06-17", + "updated": "2026-06-26", + "authors": [ + "Siyuan Pang", + "Yepeng Yao", + "Zhengwei Jiang", + "Zijing Fan", + "Haozhe Li", + "Baoxu Liu" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "rag-agent" + ], + "arxiv_id": "2606.19063", + "source": "arxiv", + "source_id": "arxiv:2606.19063", + "pdf_url": "https://arxiv.org/pdf/2606.19063", + "primary_query": "agentic-ai" + }, + { + "id": "2606.19409", + "title": "OpenRath: Session-Centered Runtime State for Agent Systems", + "url": "https://arxiv.org/abs/2606.19409", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Fukang Wen", + "Zhijie Wang", + "Ruilin Xu" + ], + "categories": [ + "cs.SE", + "cs.PL" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.19409", + "source": "arxiv", + "source_id": "arxiv:2606.19409", + "pdf_url": "https://arxiv.org/pdf/2606.19409", + "primary_query": "agent-memory" + }, + { + "id": "2606.19613", + "title": "StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns", + "url": "https://arxiv.org/abs/2606.19613", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Vlad Sobal", + "Shuo Yang", + "Yuting Zhang", + "Wei Xia", + "Stefano Soatto" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.19613", + "source": "arxiv", + "source_id": "arxiv:2606.19613", + "pdf_url": "https://arxiv.org/pdf/2606.19613", + "primary_query": "coding-agent" + }, + { + "id": "2606.20717", + "title": "MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents", + "url": "https://arxiv.org/abs/2606.20717", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Xuelong Dai", + "Jianyu Ma", + "Boyang Ma", + "Biwei Yan", + "Yijun Yang", + "Yue Zhang" + ], + "categories": [ + "cs.CV", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.20717", + "source": "arxiv", + "source_id": "arxiv:2606.20717", + "pdf_url": "https://arxiv.org/pdf/2606.20717", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.16111", + "title": "Towards Pareto-Optimal Tool-Integrated Agents with Pareto Ranking Policy Optimization", + "url": "https://arxiv.org/abs/2606.16111", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Junyi Li", + "Xiaowei Qian", + "Yingyi Zhang", + "Wenlin Zhang", + "Guojing Li", + "Sheng Zhang", + "Xiao Han", + "Yichao Wang", + "Xiangyu Zhao" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-safety", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent", + "tool-use" + ], + "arxiv_id": "2606.16111", + "source": "arxiv", + "source_id": "arxiv:2606.16111", + "pdf_url": "https://arxiv.org/pdf/2606.16111", + "primary_query": "language-agent" + }, + { + "id": "2606.16748", + "title": "MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.16748", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Lawrence Keunho Jang", + "Andrew Keunwoo Jang", + "Jing Yu Koh", + "Ruslan Salakhutdinov" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "web-gui-agent" + ], + "arxiv_id": "2606.16748", + "source": "arxiv", + "source_id": "arxiv:2606.16748", + "pdf_url": "https://arxiv.org/pdf/2606.16748", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.15591", + "title": "Agentic Retrieval and Reinforcement Learned Equation Chains: A Controlled Generation Framework for Complex and Novel Physics Word Problems", + "url": "https://arxiv.org/abs/2606.15591", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Tirthankar Mittra" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.15591", + "source": "arxiv", + "source_id": "arxiv:2606.15591", + "pdf_url": "https://arxiv.org/pdf/2606.15591", + "primary_query": "rag-agent" + }, + { + "id": "2606.15152", + "title": "Can Agents Read the Room? Benchmarking Visual Social Intelligence in Multimodal Simulation", + "url": "https://arxiv.org/abs/2606.15152", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Shijun Wan", + "Xuehai Wu", + "Jiwen Zhang", + "Siyuan Wang", + "Zhongyu Wei" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.15152", + "source": "arxiv", + "source_id": "arxiv:2606.15152", + "pdf_url": "https://arxiv.org/pdf/2606.15152", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.15242", + "title": "Benign in Isolation, Harmful in Composition: Security Risks in Agent Skill Ecosystems", + "url": "https://arxiv.org/abs/2606.15242", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Yi Xie", + "Jiawei Du", + "Yu Cheng", + "Jiuan Zhou", + "Zhaoxia Yin" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.15242", + "source": "arxiv", + "source_id": "arxiv:2606.15242", + "pdf_url": "https://arxiv.org/pdf/2606.15242", + "primary_query": "planning-agent" + }, + { + "id": "2606.14502", + "title": "From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI", + "url": "https://arxiv.org/abs/2606.14502", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Yongheng Zhang", + "Ziang Liu", + "Jiaxuan Zhu", + "Shuai Wang", + "Xiangqi Chen", + "Haojing Huang", + "Jiayi Kuang", + "Siyu Chen", + "Ao Shen", + "Hao Wu", + "Qiufeng Wang", + "Qian-Wen Zhang", + "Junnan Dong", + "Wenhao Jiang", + "Ying Shen", + "Hai-Tao Zheng", + "Yinghui Li", + "Di Yin", + "Xing Sun", + "Philip S. Yu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.14502", + "source": "arxiv", + "source_id": "arxiv:2606.14502", + "pdf_url": "https://arxiv.org/pdf/2606.14502", + "primary_query": "tool-use" + }, + { + "id": "2606.15017", + "title": "Are Online Skill and Memory Modules Always Worth Their Tokens? A Budget-Constrained Study of Web Agents", + "url": "https://arxiv.org/abs/2606.15017", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Sina Hajimiri", + "Masih Aminbeidokhti", + "Jose Dolz", + "Ismail Ben Ayed", + "Issam H. Laradji", + "Spandana Gella", + "Nicolas Gontier" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.15017", + "source": "arxiv", + "source_id": "arxiv:2606.15017", + "pdf_url": "https://arxiv.org/pdf/2606.15017", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.14574", + "title": "SIMMER: Benchmarking Latent Failures in LLM Executable Planning with a World Model", + "url": "https://arxiv.org/abs/2606.14574", + "published": "2026-06-12", + "updated": "2026-06-12", + "authors": [ + "Xiaoxin Lu", + "Ranran Haoran Zhang", + "Rui Zhang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.14574", + "source": "arxiv", + "source_id": "arxiv:2606.14574", + "pdf_url": "https://arxiv.org/pdf/2606.14574", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.13317", + "title": "SkillCAT: Contrastive Assessment and Topology-Aware Skill Self-Evolution for LLM Agents", + "url": "https://arxiv.org/abs/2606.13317", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Kunfeng Chen", + "Qihuang Zhong", + "Juhua Liu", + "Bo Du" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.13317", + "source": "arxiv", + "source_id": "arxiv:2606.13317", + "pdf_url": "https://arxiv.org/pdf/2606.13317", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.14805", + "title": "Knowledge-Based Zero-Replay Debugging of Multi-Agent LLM Traces", + "url": "https://arxiv.org/abs/2606.14805", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Dong Ho Kang", + "Hyeonjeong Cha", + "Daein Weon" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "memory", + "multi-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.14805", + "source": "arxiv", + "source_id": "arxiv:2606.14805", + "pdf_url": "https://arxiv.org/pdf/2606.14805", + "primary_query": "tool-use" + }, + { + "id": "2606.13602", + "title": "EpiBench: Verifiable Evaluation of AI Agents on Epigenomics Analysis", + "url": "https://arxiv.org/abs/2606.13602", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Harihara Muralidharan", + "Reema Baskar", + "Soo Hee Lee", + "Tim Proctor", + "Kenny Workman" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.13602", + "source": "arxiv", + "source_id": "arxiv:2606.13602", + "pdf_url": "https://arxiv.org/pdf/2606.13602", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.13192", + "title": "Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach", + "url": "https://arxiv.org/abs/2606.13192", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Ruichao Mao", + "Zhou Fang", + "Teng Guo", + "Hao Yang", + "Yaping Li", + "Shaohua Peng", + "Maji Huang", + "Xiaoyu Lin", + "Shuoyang Liu", + "Xuepeng Li", + "Yuyu Zhang", + "Hai Rao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.13192", + "source": "arxiv", + "source_id": "arxiv:2606.13192", + "pdf_url": "https://arxiv.org/pdf/2606.13192", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.12674", + "title": "Evoflux: Inference-Time Evolution of Executable Tool Workflows for Compact Agents", + "url": "https://arxiv.org/abs/2606.12674", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Kushal Raj Bhandari", + "Ling Yue", + "Ching-Yun Ko", + "Dhaval Patel", + "Shaowu Pan", + "Pin-Yu Chen", + "Jianxi Gao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling", + "tool-use" + ], + "arxiv_id": "2606.12674", + "source": "arxiv", + "source_id": "arxiv:2606.12674", + "pdf_url": "https://arxiv.org/pdf/2606.12674", + "primary_query": "function-calling" + }, + { + "id": "2606.12384", + "title": "APPO: Agentic Procedural Policy Optimization", + "url": "https://arxiv.org/abs/2606.12384", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Xucong Wang", + "Ziyu Ma", + "Yong Wang", + "Yuxiang Ji", + "Shidong Yang", + "Guanhua Chen", + "Pengkun Wang", + "Xiangxiang Chu" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.12384", + "source": "arxiv", + "source_id": "arxiv:2606.12384", + "pdf_url": "https://arxiv.org/pdf/2606.12384", + "primary_query": "tool-use" + }, + { + "id": "2606.12563", + "title": "Arbor: Tree Search as a Cognition Layer for Autonomous Agents", + "url": "https://arxiv.org/abs/2606.12563", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Neha Prakriya", + "Chaojun Hou", + "Zheng Gong", + "Huasha Zhao", + "Xi Zhao", + "Mou Li", + "Zhenyu Gu", + "Emad Barsoum" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.12563", + "source": "arxiv", + "source_id": "arxiv:2606.12563", + "pdf_url": "https://arxiv.org/pdf/2606.12563", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.11079", + "title": "VISTA: A Versatile Interactive User Simulation Toolkit for Agent Evaluation", + "url": "https://arxiv.org/abs/2606.11079", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yunan Lu", + "Ryan Shea", + "Yusen Zhang", + "Zhou Yu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.11079", + "source": "arxiv", + "source_id": "arxiv:2606.11079", + "pdf_url": "https://arxiv.org/pdf/2606.11079", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.10742", + "title": "MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents", + "url": "https://arxiv.org/abs/2606.10742", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yv Zhang", + "Hao Sun", + "Hao Fang", + "Kuofeng Gao", + "Fan Mo", + "Bin Chen", + "Shu-Tao Xia", + "Yaowei Wang" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.10742", + "source": "arxiv", + "source_id": "arxiv:2606.10742", + "pdf_url": "https://arxiv.org/pdf/2606.10742", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09774", + "title": "Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters", + "url": "https://arxiv.org/abs/2606.09774", + "published": "2026-06-08", + "updated": "2026-06-25", + "authors": [ + "Matthew Ho", + "Brian Liu", + "Jixuan Chen", + "Audrey Wang", + "Lianhui Qin" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "rag", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.09774", + "source": "arxiv", + "source_id": "arxiv:2606.09774", + "pdf_url": "https://arxiv.org/pdf/2606.09774", + "primary_query": "agent-memory" + }, + { + "id": "2606.09198", + "title": "MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation", + "url": "https://arxiv.org/abs/2606.09198", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Yongrui Liu", + "Deyi Xiong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.09198", + "source": "arxiv", + "source_id": "arxiv:2606.09198", + "pdf_url": "https://arxiv.org/pdf/2606.09198", + "primary_query": "agent-memory" + }, + { + "id": "2606.08790", + "title": "RAILS: Verification-Native Clearing For Agentic Commerce", + "url": "https://arxiv.org/abs/2606.08790", + "published": "2026-06-07", + "updated": "2026-06-07", + "authors": [ + "Adrian de Valois-Franklin", + "Alex Bogdan" + ], + "categories": [ + "cs.AI", + "cs.CR", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.08790", + "source": "arxiv", + "source_id": "arxiv:2606.08790", + "pdf_url": "https://arxiv.org/pdf/2606.08790", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.08625", + "title": "From Holistic Evaluation to Structured Criteria: Rubrics Across the Evolving LLM Landscape", + "url": "https://arxiv.org/abs/2606.08625", + "published": "2026-06-07", + "updated": "2026-07-01", + "authors": [ + "Hao Chen", + "Ziyu Han", + "Yukun Yan", + "Qingfu Zhu", + "Maosong Sun", + "Wanxiang Che" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.08625", + "source": "arxiv", + "source_id": "arxiv:2606.08625", + "pdf_url": "https://arxiv.org/pdf/2606.08625", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.07379", + "title": "Do Coding Agents Deceive Us? Detecting and Preventing Cheating via Capped Evaluation with Randomized Tests", + "url": "https://arxiv.org/abs/2606.07379", + "published": "2026-06-05", + "updated": "2026-06-08", + "authors": [ + "Thanawat Lodkaew", + "Johannes Ackermann", + "Soichiro Nishimori", + "Nontawat Charoenphakdee", + "Masashi Sugiyama", + "Takashi Ishida" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL", + "stat.ME" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.07379", + "source": "arxiv", + "source_id": "arxiv:2606.07379", + "pdf_url": "https://arxiv.org/pdf/2606.07379", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.05548", + "title": "ADK Arena: Evaluating Agent Development Kits via LLM-as-a-Developer", + "url": "https://arxiv.org/abs/2606.05548", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Jintao Huang", + "Xiaomin Li", + "Gaurav Mittal", + "Yu Hu" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm" + ], + "arxiv_id": "2606.05548", + "source": "arxiv", + "source_id": "arxiv:2606.05548", + "pdf_url": "https://arxiv.org/pdf/2606.05548", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.06473", + "title": "MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery", + "url": "https://arxiv.org/abs/2606.06473", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Shangheng Du", + "Xiangchao Yan", + "Jinxin Shi", + "Zongsheng Cao", + "Shiyang Feng", + "Zichen Liang", + "Boyuan Sun", + "Tianshuo Peng", + "Yifan Zhou", + "Xin Li", + "Jie Zhou", + "Liang He", + "Bo Zhang", + "Lei Bai" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "multi-agent", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.06473", + "source": "arxiv", + "source_id": "arxiv:2606.06473", + "pdf_url": "https://arxiv.org/pdf/2606.06473", + "primary_query": "planning-agent" + }, + { + "id": "2606.06462", + "title": "Benchmark Everything Everywhere All at Once", + "url": "https://arxiv.org/abs/2606.06462", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Shiyun Xiong", + "Dongming Wu", + "Peiwen Sun", + "Yuang Ai", + "Bokang Yang", + "Wencheng Han", + "Xiao-Hui Li", + "Xiangyu Yue" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.06462", + "source": "arxiv", + "source_id": "arxiv:2606.06462", + "pdf_url": "https://arxiv.org/pdf/2606.06462", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.05263", + "title": "Policy-Conditioned Counterfactual Credit for Verifiable Reinforcement Learning of Long-Horizon Language Agents", + "url": "https://arxiv.org/abs/2606.05263", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Renwei Meng" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.05263", + "source": "arxiv", + "source_id": "arxiv:2606.05263", + "pdf_url": "https://arxiv.org/pdf/2606.05263", + "primary_query": "language-agent" + }, + { + "id": "2606.04628", + "title": "RAMPART: Registry-based Agentic Memory with Priority-Aware Runtime Transformation", + "url": "https://arxiv.org/abs/2606.04628", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Nikodem Tomczak" + ], + "categories": [ + "cs.CL", + "cs.MA" + ], + "topics": [ + "memory" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.04628", + "source": "arxiv", + "source_id": "arxiv:2606.04628", + "pdf_url": "https://arxiv.org/pdf/2606.04628", + "primary_query": "agent-memory" + }, + { + "id": "2606.05414", + "title": "When Evidence is Sparse: Weakly Supervised Early Failure Alerting in Dialogs and LLM-Agent Trajectories", + "url": "https://arxiv.org/abs/2606.05414", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Avinash Baidya", + "Xinran Liang", + "Ruocheng Guo", + "Xiang Gao", + "Kamalika Das" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.HC", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.05414", + "source": "arxiv", + "source_id": "arxiv:2606.05414", + "pdf_url": "https://arxiv.org/pdf/2606.05414", + "primary_query": "planning-agent" + }, + { + "id": "2606.03329", + "title": "InfoMem: Training Long-Context Memory Agents with Answer-Conditioned Information Gain", + "url": "https://arxiv.org/abs/2606.03329", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Tiancheng Han", + "Yong Li", + "Wuzhou Yu", + "Qiaosheng Zhang", + "Wenqi Shao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.03329", + "source": "arxiv", + "source_id": "arxiv:2606.03329", + "pdf_url": "https://arxiv.org/pdf/2606.03329", + "primary_query": "agent-memory" + }, + { + "id": "2606.02965", + "title": "What Benchmarks Don't Measure: The Case for Evaluating Abstention Competence in Autonomous Agents", + "url": "https://arxiv.org/abs/2606.02965", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Victor Ojewale", + "Suresh Venkatasubramanian" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.02965", + "source": "arxiv", + "source_id": "arxiv:2606.02965", + "pdf_url": "https://arxiv.org/pdf/2606.02965", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.02404", + "title": "K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts", + "url": "https://arxiv.org/abs/2606.02404", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Nahyun Lee", + "Dongkeun Yoon", + "Guijin Son", + "Geewook Kim", + "Dayoon Ko", + "Jeonghun Park", + "Haneul Yoo", + "Jaewon Cho", + "Junghun Park", + "Changyoon Lee", + "Kyochul Jang", + "Jaeyeon Kim", + "Eunsu Kim", + "Woojin Cho", + "Seungone Kim" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.02404", + "source": "arxiv", + "source_id": "arxiv:2606.02404", + "pdf_url": "https://arxiv.org/pdf/2606.02404", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.09863", + "title": "From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents", + "url": "https://arxiv.org/abs/2606.09863", + "published": "2026-06-01", + "updated": "2026-06-01", + "authors": [ + "Laksh Advani" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.09863", + "source": "arxiv", + "source_id": "arxiv:2606.09863", + "pdf_url": "https://arxiv.org/pdf/2606.09863", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.00611", + "title": "TRACE: Trajectory Risk-Aware Compression for Long-Horizon Agent Safety", + "url": "https://arxiv.org/abs/2606.00611", + "published": "2026-05-30", + "updated": "2026-05-30", + "authors": [ + "Zhepei Hong", + "Lin Wang", + "Liting Li", + "Haokai Ma", + "Junfeng Fang", + "Fei Shen", + "Dan Zhang", + "Xiang Wang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.00611", + "source": "arxiv", + "source_id": "arxiv:2606.00611", + "pdf_url": "https://arxiv.org/pdf/2606.00611", + "primary_query": "agent-safety" + }, + { + "id": "2606.00198", + "title": "BAGEN: Are LLM Agents Budget-Aware?", + "url": "https://arxiv.org/abs/2606.00198", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Yuxiang Lin", + "Zihan Wang", + "Mengyang Liu", + "Yuxuan Shan", + "Longju Bai", + "Junyao Zhang", + "Xing Jin", + "Boshan Chen", + "Jinyan Su", + "Xingyao Wang", + "Jiaxin Pei", + "Manling Li" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.00198", + "source": "arxiv", + "source_id": "arxiv:2606.00198", + "pdf_url": "https://arxiv.org/pdf/2606.00198", + "primary_query": "planning-agent" + }, + { + "id": "2605.29790", + "title": "Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems", + "url": "https://arxiv.org/abs/2605.29790", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Zhezheng Hao", + "Tianfu Wang", + "Huanshuo Dong", + "Ziyan Liu", + "Hong Wang", + "Xiankun Lin", + "Qiang Lin", + "Can Wang", + "Hande Dong", + "Jiawei Chen" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.29790", + "source": "arxiv", + "source_id": "arxiv:2605.29790", + "pdf_url": "https://arxiv.org/pdf/2605.29790", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.29640", + "title": "VikingMem: A Memory Base Management System for Stateful LLM-based Applications", + "url": "https://arxiv.org/abs/2605.29640", + "published": "2026-05-28", + "updated": "2026-06-12", + "authors": [ + "Jiajie Fu", + "Junwen Chen", + "Mengzhao Wang", + "Aoxiang He", + "Maojia Sheng", + "Xiangyu Ke", + "Yifan Zhu", + "Yunjun Gao" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.29640", + "source": "arxiv", + "source_id": "arxiv:2605.29640", + "pdf_url": "https://arxiv.org/pdf/2605.29640", + "primary_query": "agent-memory" + }, + { + "id": "2605.29801", + "title": "AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security", + "url": "https://arxiv.org/abs/2605.29801", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Dongrui Liu", + "Yu Li", + "Zhonghao Yang", + "Peng Wang", + "Guanxu Chen", + "Yuejin Xie", + "Qinghua Mao", + "Wanying Qu", + "Yanxu Zhu", + "Tianyi Zhou", + "Leitao Yuan", + "Zhijie Zheng", + "Qihao Lin", + "Yimin Wang", + "Haoyu Luo", + "Shuai Shao", + "Chen Qian", + "Qingyu Liu", + "Ling Tang", + "Ruiyang Qin", + "Qihan Ren", + "Junxiao Yang", + "Kun Wang", + "Zhiheng Xi", + "Linfeng Zhang", + "Ranjie Duan", + "Bo Zhang", + "Wenjie Wang", + "Wen Shen", + "Qiaosheng Zhang", + "Yan Teng", + "Chaochao Lu", + "Rui Mei", + "Man Li", + "Jialing Tao", + "Xi Lin", + "Tianhang Zheng", + "Yong Liu", + "Quanshi Zhang", + "Lei Zhu", + "Xingjun Ma", + "Junhua Liu", + "Hui Xue", + "Xiaoxiang Zuo", + "Xiangnan He", + "Chao Shen", + "Xianglong Liu", + "Minlie Huang", + "Jing Shao", + "Xia Hu" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.CR", + "cs.CV", + "cs.LG" + ], + "topics": [ + "agent-safety", + "computer-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.29801", + "source": "arxiv", + "source_id": "arxiv:2605.29801", + "pdf_url": "https://arxiv.org/pdf/2605.29801", + "primary_query": "agent-safety" + }, + { + "id": "2605.30407", + "title": "Exploring Autonomous Agentic Data Engineering for Model Specialization", + "url": "https://arxiv.org/abs/2605.30407", + "published": "2026-05-28", + "updated": "2026-06-08", + "authors": [ + "Yujie Luo", + "Xiangyuan Ru", + "Jingsheng Zheng", + "Jingjing Wang", + "Yuqi Zhu", + "Jintian Zhang", + "Runnan Fang", + "Kewei Xu", + "Ye Liu", + "Zheng Wei", + "Jiang Bian", + "Zang Li", + "Shumin Deng" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.30407", + "source": "arxiv", + "source_id": "arxiv:2605.30407", + "pdf_url": "https://arxiv.org/pdf/2605.30407", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.27825", + "title": "MRMMIA: Membership Inference Attacks on Memory in Chat Agents", + "url": "https://arxiv.org/abs/2605.27825", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Kai Chen", + "Yan Pang", + "Tianhao Wang" + ], + "categories": [ + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.27825", + "source": "arxiv", + "source_id": "arxiv:2605.27825", + "pdf_url": "https://arxiv.org/pdf/2605.27825", + "primary_query": "agent-memory" + }, + { + "id": "2605.28175", + "title": "Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation", + "url": "https://arxiv.org/abs/2605.28175", + "published": "2026-05-27", + "updated": "2026-05-29", + "authors": [ + "Shijie Wang", + "Chengyi Liu", + "Yujuan Ding", + "Shanru Lin", + "See-Kiong Ng", + "Xu Xin", + "Wenqi Fan" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.28175", + "source": "arxiv", + "source_id": "arxiv:2605.28175", + "pdf_url": "https://arxiv.org/pdf/2605.28175", + "primary_query": "rag-agent" + }, + { + "id": "2605.27935", + "title": "Do Agents Think Deeper? A Mechanistic Investigation of Layer-Wise Dynamics in Sequential Planning", + "url": "https://arxiv.org/abs/2605.27935", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Zhenyu Cui", + "Xiangzhong Luo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm", + "planning-agent" + ], + "arxiv_id": "2605.27935", + "source": "arxiv", + "source_id": "arxiv:2605.27935", + "pdf_url": "https://arxiv.org/pdf/2605.27935", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.25920", + "title": "Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning", + "url": "https://arxiv.org/abs/2605.25920", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Wei Fan", + "Yining Zhou", + "Mufan Zhang", + "Yanbing Weng", + "Yiran HU", + "Tianshi Zheng", + "Baixuan Xu", + "Chunyang Li", + "Jianhui Yang", + "Haoran Li", + "Yangqiu Song" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.25920", + "source": "arxiv", + "source_id": "arxiv:2605.25920", + "pdf_url": "https://arxiv.org/pdf/2605.25920", + "primary_query": "rag-agent" + }, + { + "id": "2605.25393", + "title": "Decision-Making with Lightweight Confidence-Aware Language Model for Autonomous Driving", + "url": "https://arxiv.org/abs/2605.25393", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Ruoyu Yao", + "Ruiguo Zhong", + "Pei Liu", + "Mingxing Peng", + "Rui Yang", + "Jun Ma" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.25393", + "source": "arxiv", + "source_id": "arxiv:2605.25393", + "pdf_url": "https://arxiv.org/pdf/2605.25393", + "primary_query": "rag-agent" + }, + { + "id": "2605.25310", + "title": "Tool-Call Dependency Structure is Linearly Decodable in LLM Agent Residual Streams", + "url": "https://arxiv.org/abs/2605.25310", + "published": "2026-05-25", + "updated": "2026-05-25", + "authors": [ + "Tianda Sun", + "Dimitar Kazakov" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.25310", + "source": "arxiv", + "source_id": "arxiv:2605.25310", + "pdf_url": "https://arxiv.org/pdf/2605.25310", + "primary_query": "planning-agent" + }, + { + "id": "2605.24309", + "title": "Reframing LLM Agent Security as an Agent-Human Interaction Problem", + "url": "https://arxiv.org/abs/2605.24309", + "published": "2026-05-23", + "updated": "2026-05-23", + "authors": [ + "Peiran Wang", + "Ying Li", + "Yuan Tian" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.24309", + "source": "arxiv", + "source_id": "arxiv:2605.24309", + "pdf_url": "https://arxiv.org/pdf/2605.24309", + "primary_query": "agent-safety" + }, + { + "id": "2605.19604", + "title": "Formal Skill: Programmable Runtime Skills for Efficient and Accurate LLM Agents", + "url": "https://arxiv.org/abs/2605.19604", + "published": "2026-05-19", + "updated": "2026-05-19", + "authors": [ + "Xi Zhang", + "Meijun Gao", + "Yuntian Zhao", + "Xinyu Tan", + "Yilun Yao", + "Feiyu Wang", + "Yanshu Wang", + "Dingsiyi", + "Tong Yang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.19604", + "source": "arxiv", + "source_id": "arxiv:2605.19604", + "pdf_url": "https://arxiv.org/pdf/2605.19604", + "primary_query": "function-calling" + }, + { + "id": "2605.18502", + "title": "The distance-based formation controller design for multi-agent systems in port-Hamiltonian form", + "url": "https://arxiv.org/abs/2605.18502", + "published": "2026-05-18", + "updated": "2026-05-18", + "authors": [ + "Jingyi Zhao", + "Yongxin Wu", + "Héctor García de Marina", + "Yuhu Wu", + "Yann Le Gorrec" + ], + "categories": [ + "math.OC" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.18502", + "source": "arxiv", + "source_id": "arxiv:2605.18502", + "pdf_url": "https://arxiv.org/pdf/2605.18502", + "primary_query": "agent-safety" + }, + { + "id": "2605.15701", + "title": "H-Mem: A Novel Memory Mechanism for Evolving and Retrieving Agent Memory via a Hybrid Structure", + "url": "https://arxiv.org/abs/2605.15701", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Jiawei Yu", + "Yixiang Fang", + "Xilin Liu", + "Yuchi Ma" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.15701", + "source": "arxiv", + "source_id": "arxiv:2605.15701", + "pdf_url": "https://arxiv.org/pdf/2605.15701", + "primary_query": "agent-memory" + }, + { + "id": "2605.15625", + "title": "ColPackAgent: Agent-Skill-Guided Hard-Particle Monte Carlo Workflows for Colloidal Packing", + "url": "https://arxiv.org/abs/2605.15625", + "published": "2026-05-15", + "updated": "2026-05-15", + "authors": [ + "Lijie Ding", + "Changwoo Do" + ], + "categories": [ + "cs.AI", + "cond-mat.soft" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.15625", + "source": "arxiv", + "source_id": "arxiv:2605.15625", + "pdf_url": "https://arxiv.org/pdf/2605.15625", + "primary_query": "planning-agent" + }, + { + "id": "2605.14290", + "title": "Web Agents Should Adopt the Plan-Then-Execute Paradigm", + "url": "https://arxiv.org/abs/2605.14290", + "published": "2026-05-14", + "updated": "2026-05-14", + "authors": [ + "Julien Piet", + "Annabella Chow", + "Yiwei Hou", + "Muxi Lyu", + "Sylvie Venuto", + "Jinhao Zhu", + "Raluca Ada Popa", + "David Wagner" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.14290", + "source": "arxiv", + "source_id": "arxiv:2605.14290", + "pdf_url": "https://arxiv.org/pdf/2605.14290", + "primary_query": "planning-agent" + }, + { + "id": "2605.13716", + "title": "SkillOps: Managing LLM Agent Skill Libraries as Self-Maintaining Software Ecosystems", + "url": "https://arxiv.org/abs/2605.13716", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Hongji Pu", + "Xinyuan Song", + "Liang Zhao" + ], + "categories": [ + "cs.SE", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.13716", + "source": "arxiv", + "source_id": "arxiv:2605.13716", + "pdf_url": "https://arxiv.org/pdf/2605.13716", + "primary_query": "planning-agent" + }, + { + "id": "2605.13618", + "title": "OpenAaaS: An Open Agent-as-a-Service Framework for Distributed Materials-Informatics Research", + "url": "https://arxiv.org/abs/2605.13618", + "published": "2026-05-13", + "updated": "2026-05-13", + "authors": [ + "Peng Kang", + "Bixuan Li", + "Xiaoya Huang", + "Shuo Shi", + "Weiqiao Zhou", + "Zhen Li", + "Yu Liu", + "Lei Zheng" + ], + "categories": [ + "cond-mat.mtrl-sci", + "cs.AI" + ], + "topics": [ + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.13618", + "source": "arxiv", + "source_id": "arxiv:2605.13618", + "pdf_url": "https://arxiv.org/pdf/2605.13618", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.07112", + "title": "Switchcraft: AI Model Router for Agentic Tool Calling", + "url": "https://arxiv.org/abs/2605.07112", + "published": "2026-05-08", + "updated": "2026-05-08", + "authors": [ + "Sharad Agarwal", + "Pooria Namyar", + "Alec Wolman", + "Rahul Ambavat", + "Ankur Gupta", + "Qizheng Zhang" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.07112", + "source": "arxiv", + "source_id": "arxiv:2605.07112", + "pdf_url": "https://arxiv.org/pdf/2605.07112", + "primary_query": "function-calling" + }, + { + "id": "2605.06992", + "title": "Why Does Agentic Safety Fail to Generalize Across Tasks?", + "url": "https://arxiv.org/abs/2605.06992", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Yonatan Slutzky", + "Yotam Alexander", + "Tomer Slor", + "Yoav Nagel", + "Nadav Cohen" + ], + "categories": [ + "cs.LG", + "stat.ML" + ], + "topics": [ + "agent-safety", + "embodied-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.06992", + "source": "arxiv", + "source_id": "arxiv:2605.06992", + "pdf_url": "https://arxiv.org/pdf/2605.06992", + "primary_query": "agent-safety" + }, + { + "id": "2605.06957", + "title": "Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents", + "url": "https://arxiv.org/abs/2605.06957", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Shirin Sohrabi", + "Haritha Ananthakrishnan", + "Harsha Kokel", + "Kavitha Srinivas", + "Michael Katz" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.06957", + "source": "arxiv", + "source_id": "arxiv:2605.06957", + "pdf_url": "https://arxiv.org/pdf/2605.06957", + "primary_query": "planning-agent" + }, + { + "id": "2605.06737", + "title": "A Self-Healing Framework for Reliable LLM-Based Autonomous Agents", + "url": "https://arxiv.org/abs/2605.06737", + "published": "2026-05-07", + "updated": "2026-05-07", + "authors": [ + "Cheonsu Jeong", + "Younggun Shin" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.06737", + "source": "arxiv", + "source_id": "arxiv:2605.06737", + "pdf_url": "https://arxiv.org/pdf/2605.06737", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.27464", + "title": "Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study", + "url": "https://arxiv.org/abs/2604.27464", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Luyao Xu", + "Xiang Chen" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety", + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27464", + "source": "arxiv", + "source_id": "arxiv:2604.27464", + "pdf_url": "https://arxiv.org/pdf/2604.27464", + "primary_query": "agent-safety" + }, + { + "id": "2604.28157", + "title": "FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption", + "url": "https://arxiv.org/abs/2604.28157", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Yanting Wang", + "Chenlong Yin", + "Ying Chen", + "Jinyuan Jia" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.28157", + "source": "arxiv", + "source_id": "arxiv:2604.28157", + "pdf_url": "https://arxiv.org/pdf/2604.28157", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.27859", + "title": "Rethinking Agentic Reinforcement Learning In Large Language Models", + "url": "https://arxiv.org/abs/2604.27859", + "published": "2026-04-30", + "updated": "2026-05-15", + "authors": [ + "Fangming Cui", + "Ruixiao Zhu", + "Cheng Fang", + "Sunan Li", + "Jiahong Li" + ], + "categories": [ + "cs.AI", + "cs.ET" + ], + "topics": [ + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27859", + "source": "arxiv", + "source_id": "arxiv:2604.27859", + "pdf_url": "https://arxiv.org/pdf/2604.27859", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.20573", + "title": "AONA: A Comprehensive Architecture and Workflow Design for Global Agentic Collaboration", + "url": "https://arxiv.org/abs/2606.20573", + "published": "2026-04-30", + "updated": "2026-04-30", + "authors": [ + "Jinliang Xu", + "Runkai Zhu", + "Bingqi Li", + "Fanjie Nie", + "Jin Li", + "Jiagui Xie" + ], + "categories": [ + "cs.NI", + "cs.MA" + ], + "topics": [ + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.20573", + "source": "arxiv", + "source_id": "arxiv:2606.20573", + "pdf_url": "https://arxiv.org/pdf/2606.20573", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.25555", + "title": "From CRUD to Autonomous Agents: Formal Validation and Zero-Trust Security for Semantic Gateways in AI-Native Enterprise Systems", + "url": "https://arxiv.org/abs/2604.25555", + "published": "2026-04-28", + "updated": "2026-04-28", + "authors": [ + "Ignacio Peyrano" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.25555", + "source": "arxiv", + "source_id": "arxiv:2604.25555", + "pdf_url": "https://arxiv.org/pdf/2604.25555", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.21190", + "title": "SpatiO: Adaptive Test-Time Orchestration of Vision-Language Agents for Spatial Reasoning", + "url": "https://arxiv.org/abs/2604.21190", + "published": "2026-04-23", + "updated": "2026-06-27", + "authors": [ + "Chan Yeong Hwang", + "Miso Choi", + "Sunghyun On", + "Jinkyu Kim", + "Jungbeom Lee" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.21190", + "source": "arxiv", + "source_id": "arxiv:2604.21190", + "pdf_url": "https://arxiv.org/pdf/2604.21190", + "primary_query": "language-agent" + }, + { + "id": "2604.16706", + "title": "Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench", + "url": "https://arxiv.org/abs/2604.16706", + "published": "2026-04-17", + "updated": "2026-04-17", + "authors": [ + "Bhaskar Gurram" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.16706", + "source": "arxiv", + "source_id": "arxiv:2604.16706", + "pdf_url": "https://arxiv.org/pdf/2604.16706", + "primary_query": "language-agent" + }, + { + "id": "2604.15579", + "title": "Don't Make Models Guess Security and Safety: Symbolic Guardrails for Domain-Specific AI Agents", + "url": "https://arxiv.org/abs/2604.15579", + "published": "2026-04-16", + "updated": "2026-07-05", + "authors": [ + "Yining Hong", + "Yining She", + "Eunsuk Kang", + "Christopher S. Timperley", + "Christian Kästner" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.15579", + "source": "arxiv", + "source_id": "arxiv:2604.15579", + "pdf_url": "https://arxiv.org/pdf/2604.15579", + "primary_query": "agent-safety" + }, + { + "id": "2604.15415", + "title": "HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?", + "url": "https://arxiv.org/abs/2604.15415", + "published": "2026-04-16", + "updated": "2026-04-16", + "authors": [ + "Yukun Jiang", + "Yage Zhang", + "Michael Backes", + "Xinyue Shen", + "Yang Zhang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.15415", + "source": "arxiv", + "source_id": "arxiv:2604.15415", + "pdf_url": "https://arxiv.org/pdf/2604.15415", + "primary_query": "agent-safety" + }, + { + "id": "2604.14399", + "title": "SpaceMind: A Modular and Self-Evolving Embodied Vision-Language Agent Framework for Autonomous On-orbit Servicing", + "url": "https://arxiv.org/abs/2604.14399", + "published": "2026-04-15", + "updated": "2026-04-15", + "authors": [ + "Aodi Wu", + "Haodong Han", + "Xubo Luo", + "Ruisuo Wang", + "Shan He", + "Xue Wan" + ], + "categories": [ + "cs.RO", + "cs.AI", + "eess.SY" + ], + "topics": [ + "embodied-agent", + "reasoning", + "tool-use", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2604.14399", + "source": "arxiv", + "source_id": "arxiv:2604.14399", + "pdf_url": "https://arxiv.org/pdf/2604.14399", + "primary_query": "language-agent" + }, + { + "id": "2604.08388", + "title": "Awakening the Sleeping Agent: Lean-Specific Agentic Data Reactivates General Tool Use in Goedel Prover", + "url": "https://arxiv.org/abs/2604.08388", + "published": "2026-04-09", + "updated": "2026-04-09", + "authors": [ + "Jui-Hui Chung", + "Hongzhou Lin", + "Lai Jiang", + "Shange Tang", + "Chi Jin" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2604.08388", + "source": "arxiv", + "source_id": "arxiv:2604.08388", + "pdf_url": "https://arxiv.org/pdf/2604.08388", + "primary_query": "function-calling" + }, + { + "id": "2604.06762", + "title": "ARuleCon: Agentic Security Rule Conversion", + "url": "https://arxiv.org/abs/2604.06762", + "published": "2026-04-08", + "updated": "2026-04-08", + "authors": [ + "Ming Xu", + "Hongtai Wang", + "Yanpei Guo", + "Zhengmin Yu", + "Weili Han", + "Hoon Wei Lim", + "Jin Song Dong", + "Jiaheng Zhang" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2604.06762", + "source": "arxiv", + "source_id": "arxiv:2604.06762", + "pdf_url": "https://arxiv.org/pdf/2604.06762", + "primary_query": "agent-safety" + }, + { + "id": "2604.02155", + "title": "Brief Is Better: Non-Monotonic Chain-of-Thought Budget Effects in Function-Calling Language Agents", + "url": "https://arxiv.org/abs/2604.02155", + "published": "2026-04-02", + "updated": "2026-04-02", + "authors": [ + "Xuan Qi" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling", + "language-agent" + ], + "arxiv_id": "2604.02155", + "source": "arxiv", + "source_id": "arxiv:2604.02155", + "pdf_url": "https://arxiv.org/pdf/2604.02155", + "primary_query": "function-calling" + }, + { + "id": "2603.27148", + "title": "SafetyDrift: Predicting When AI Agents Cross the Line Before They Actually Do", + "url": "https://arxiv.org/abs/2603.27148", + "published": "2026-03-28", + "updated": "2026-03-28", + "authors": [ + "Aditya Dhodapkar", + "Farhaan Pishori" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.27148", + "source": "arxiv", + "source_id": "arxiv:2603.27148", + "pdf_url": "https://arxiv.org/pdf/2603.27148", + "primary_query": "agent-safety" + }, + { + "id": "2603.19469", + "title": "A Framework for Formalizing LLM Agent Security", + "url": "https://arxiv.org/abs/2603.19469", + "published": "2026-03-19", + "updated": "2026-03-19", + "authors": [ + "Vincent Siu", + "Jingxuan He", + "Kyle Montgomery", + "Zhun Wang", + "Neil Gong", + "Chenguang Wang", + "Dawn Song" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "memory", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.19469", + "source": "arxiv", + "source_id": "arxiv:2603.19469", + "pdf_url": "https://arxiv.org/pdf/2603.19469", + "primary_query": "agent-safety" + }, + { + "id": "2603.11088", + "title": "The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey", + "url": "https://arxiv.org/abs/2603.11088", + "published": "2026-03-11", + "updated": "2026-03-11", + "authors": [ + "Juhee Kim", + "Xiaoyuan Liu", + "Zhun Wang", + "Shi Qiu", + "Bo Li", + "Wenbo Guo", + "Dawn Song" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.11088", + "source": "arxiv", + "source_id": "arxiv:2603.11088", + "pdf_url": "https://arxiv.org/pdf/2603.11088", + "primary_query": "agent-safety" + }, + { + "id": "2603.01438", + "title": "Enhancing Persona Following at Decoding Time via Dynamic Importance Estimation for Role-Playing Agents", + "url": "https://arxiv.org/abs/2603.01438", + "published": "2026-03-02", + "updated": "2026-03-02", + "authors": [ + "Yuxin Liu", + "Mingye Zhu", + "Siyuan Liu", + "Bo Hu", + "Lei Zhang" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "planning", + "rag", + "world-model" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.01438", + "source": "arxiv", + "source_id": "arxiv:2603.01438", + "pdf_url": "https://arxiv.org/pdf/2603.01438", + "primary_query": "language-agent" + }, + { + "id": "2602.23320", + "title": "ParamMem: Augmenting Language Agents with Parametric Reflective Memory", + "url": "https://arxiv.org/abs/2602.23320", + "published": "2026-02-26", + "updated": "2026-02-27", + "authors": [ + "Tianjun Yao", + "Yongqiang Chen", + "Yujia Zheng", + "Pan Li", + "Zhiqiang Shen", + "Kun Zhang" + ], + "categories": [ + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "memory", + "reasoning" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.23320", + "source": "arxiv", + "source_id": "arxiv:2602.23320", + "pdf_url": "https://arxiv.org/pdf/2602.23320", + "primary_query": "language-agent" + }, + { + "id": "2602.16931", + "title": "Narrow Fine-Tuning Erodes Safety Alignment in Vision-Language Agents", + "url": "https://arxiv.org/abs/2602.16931", + "published": "2026-02-18", + "updated": "2026-03-15", + "authors": [ + "Idhant Gulati", + "Shivam Raval" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.16931", + "source": "arxiv", + "source_id": "arxiv:2602.16931", + "pdf_url": "https://arxiv.org/pdf/2602.16931", + "primary_query": "language-agent" + }, + { + "id": "2602.07391", + "title": "NAAMSE: Framework for Evolutionary Security Evaluation of Agents", + "url": "https://arxiv.org/abs/2602.07391", + "published": "2026-02-07", + "updated": "2026-03-08", + "authors": [ + "Kunal Pai", + "Parth Shah", + "Harshil Patel" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.07391", + "source": "arxiv", + "source_id": "arxiv:2602.07391", + "pdf_url": "https://arxiv.org/pdf/2602.07391", + "primary_query": "agent-safety" + }, + { + "id": "2601.05467", + "title": "STELP: Secure Transpilation and Execution of LLM-Generated Programs", + "url": "https://arxiv.org/abs/2601.05467", + "published": "2026-01-09", + "updated": "2026-01-15", + "authors": [ + "Swapnil Shinde", + "Sahil Wadhwa", + "Andy Luo", + "Akshay Gupta", + "Mohammad Shahed Sorower" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.05467", + "source": "arxiv", + "source_id": "arxiv:2601.05467", + "pdf_url": "https://arxiv.org/pdf/2601.05467", + "primary_query": "function-calling" + }, + { + "id": "2510.26167", + "title": "ToolRM: Towards Agentic Tool-Use Reward Modeling", + "url": "https://arxiv.org/abs/2510.26167", + "published": "2025-10-30", + "updated": "2026-01-13", + "authors": [ + "Renhao Li", + "Jianhong Tu", + "Yang Su", + "Yantao Liu", + "Fei Huang", + "Hamid Alinejad-Rokny", + "Derek F. Wong", + "Junyang Lin", + "Min Yang" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.26167", + "source": "arxiv", + "source_id": "arxiv:2510.26167", + "pdf_url": "https://arxiv.org/pdf/2510.26167", + "primary_query": "function-calling" + }, + { + "id": "2510.22768", + "title": "Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion", + "url": "https://arxiv.org/abs/2510.22768", + "published": "2025-10-26", + "updated": "2026-06-02", + "authors": [ + "Haoyi Qiu", + "Yilun Zhou", + "Pranav Narayanan Venkit", + "Kung-Hsiang Huang", + "Jiaxin Zhang", + "Nanyun Peng", + "Chien-Sheng Wu" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "score": 14, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.22768", + "source": "arxiv", + "source_id": "arxiv:2510.22768", + "pdf_url": "https://arxiv.org/pdf/2510.22768", + "primary_query": "function-calling" + }, + { + "id": "2607.05794", + "title": "From Passive Retrieval to Active Memory Navigation: Learning to Use Memory as a Structured Action Space", + "url": "https://arxiv.org/abs/2607.05794", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Yue Xu", + "Yutao Sun", + "Yihao Liu", + "Mengyu Zhou", + "Jiayi Qiao", + "Lu Ma", + "Kai Tang", + "Wenjie Wang", + "Xiaoxi Jiang", + "Guanjun Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2607.05794", + "source": "arxiv", + "source_id": "arxiv:2607.05794", + "pdf_url": "https://arxiv.org/pdf/2607.05794", + "primary_query": "tool-use" + }, + { + "id": "2607.06341", + "title": "Harnessing Code Agents for Automatic Software Verification", + "url": "https://arxiv.org/abs/2607.06341", + "published": "2026-07-07", + "updated": "2026-07-07", + "authors": [ + "Shuangxiang Kan", + "Shuanglong Kan", + "Sebastian Ertel" + ], + "categories": [ + "cs.FL", + "cs.AI", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.06341", + "source": "arxiv", + "source_id": "arxiv:2607.06341", + "pdf_url": "https://arxiv.org/pdf/2607.06341", + "primary_query": "coding-agent" + }, + { + "id": "2607.05001", + "title": "TACTIC-KG: Toward Small Agent Teams for Cyber Threat Intelligence Knowledge Graph Construction", + "url": "https://arxiv.org/abs/2607.05001", + "published": "2026-07-06", + "updated": "2026-07-07", + "authors": [ + "Mouhamed Amine Bouchiha", + "Gregory Blanc" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.05001", + "source": "arxiv", + "source_id": "arxiv:2607.05001", + "pdf_url": "https://arxiv.org/pdf/2607.05001", + "primary_query": "llm-agent" + }, + { + "id": "2607.05666", + "title": "What Do AI Agents Actually Change? An Empirical Taxonomy of Mutation Patterns in Performance-Improving Pull Requests", + "url": "https://arxiv.org/abs/2607.05666", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Illia Dovhoshliubnyi", + "Nima Soroush", + "Ashkan Sami", + "Alexander Brownlee" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "coding-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2607.05666", + "source": "arxiv", + "source_id": "arxiv:2607.05666", + "pdf_url": "https://arxiv.org/pdf/2607.05666", + "primary_query": "ai-agent" + }, + { + "id": "2607.05518", + "title": "aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents", + "url": "https://arxiv.org/abs/2607.05518", + "published": "2026-07-06", + "updated": "2026-07-06", + "authors": [ + "Sai Varun Kodathala" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.05518", + "source": "arxiv", + "source_id": "arxiv:2607.05518", + "pdf_url": "https://arxiv.org/pdf/2607.05518", + "primary_query": "ai-agent" + }, + { + "id": "2607.04697", + "title": "AI Agent Pull Requests on GitHub: Frequency, Structure, and Merge Conflict Rates", + "url": "https://arxiv.org/abs/2607.04697", + "published": "2026-07-06", + "updated": "2026-07-07", + "authors": [ + "George Xu", + "Arjun Subramanian", + "Nithilan Karthik" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "coding-agent" + ], + "arxiv_id": "2607.04697", + "source": "arxiv", + "source_id": "arxiv:2607.04697", + "pdf_url": "https://arxiv.org/pdf/2607.04697", + "primary_query": "ai-agent" + }, + { + "id": "2607.05391", + "title": "LLM-as-a-Verifier: A General-Purpose Verification Framework", + "url": "https://arxiv.org/abs/2607.05391", + "published": "2026-07-06", + "updated": "2026-07-07", + "authors": [ + "Jacky Kwok", + "Shulu Li", + "Pranav Atreya", + "Yuejiang Liu", + "Yixing Jiang", + "Chelsea Finn", + "Marco Pavone", + "Ion Stoica", + "Azalia Mirhoseini" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA", + "cs.RO" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.05391", + "source": "arxiv", + "source_id": "arxiv:2607.05391", + "pdf_url": "https://arxiv.org/pdf/2607.05391", + "primary_query": "coding-agent" + }, + { + "id": "2607.04334", + "title": "Do GUI Agents Believe Their Eyes? Diagnosing State-Belief Reliance on Pixels versus Structure", + "url": "https://arxiv.org/abs/2607.04334", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Guijia Zhang", + "Harry Yang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2607.04334", + "source": "arxiv", + "source_id": "arxiv:2607.04334", + "pdf_url": "https://arxiv.org/pdf/2607.04334", + "primary_query": "web-gui-agent" + }, + { + "id": "2607.04394", + "title": "MechMath Agent Team: LLM Driven Agents for Mathematical Research", + "url": "https://arxiv.org/abs/2607.04394", + "published": "2026-07-05", + "updated": "2026-07-05", + "authors": [ + "Yichuan Cao", + "Ruichen Qiu", + "Junqi Liu", + "Jiaqi Wang", + "Dakai Guo", + "Ruyong Feng", + "Lihong Zhi", + "Xiao-Shan Gao" + ], + "categories": [ + "cs.AI", + "cs.SC" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2607.04394", + "source": "arxiv", + "source_id": "arxiv:2607.04394", + "pdf_url": "https://arxiv.org/pdf/2607.04394", + "primary_query": "multi-agent-llm" + }, + { + "id": "2607.02942", + "title": "A Workflow-Aware Serving Layer for Agentic Applications", + "url": "https://arxiv.org/abs/2607.02942", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Jiayi Qian", + "Zishen Wan", + "Hanchen Yang", + "Chun Tao", + "Souvik Kundu", + "Tushar Krishna" + ], + "categories": [ + "cs.DC", + "cs.MA" + ], + "topics": [ + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02942", + "source": "arxiv", + "source_id": "arxiv:2607.02942", + "pdf_url": "https://arxiv.org/pdf/2607.02942", + "primary_query": "agentic-ai" + }, + { + "id": "2607.03105", + "title": "ORBIT-Q: Dual-axis benchmarking of autonomous agents in scientific quantum programming", + "url": "https://arxiv.org/abs/2607.03105", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Shi-Xin Zhang", + "Yu-Qin Chen" + ], + "categories": [ + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.03105", + "source": "arxiv", + "source_id": "arxiv:2607.03105", + "pdf_url": "https://arxiv.org/pdf/2607.03105", + "primary_query": "coding-agent" + }, + { + "id": "2607.03316", + "title": "Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild", + "url": "https://arxiv.org/abs/2607.03316", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Hong Yi Lin", + "Mingzhao Liang", + "Kla Tantithamthavorn", + "Patanamon Thongtanunam" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-safety", + "coding-agent", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2607.03316", + "source": "arxiv", + "source_id": "arxiv:2607.03316", + "pdf_url": "https://arxiv.org/pdf/2607.03316", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.03220", + "title": "CONTRA: Red-Teaming Configurations of Personalizable Agents", + "url": "https://arxiv.org/abs/2607.03220", + "published": "2026-07-03", + "updated": "2026-07-03", + "authors": [ + "Jonathan Nöther", + "Adish Singla", + "Goran Radanovic" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2607.03220", + "source": "arxiv", + "source_id": "arxiv:2607.03220", + "pdf_url": "https://arxiv.org/pdf/2607.03220", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2607.01812", + "title": "TO-Master: an LLM-agent framework for automated topology optimization", + "url": "https://arxiv.org/abs/2607.01812", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Haoju Lin", + "Wenchang Zhang", + "Weipeng Xu", + "Xiang Li", + "Tian Xu", + "Tianju Xue" + ], + "categories": [ + "cs.CE" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01812", + "source": "arxiv", + "source_id": "arxiv:2607.01812", + "pdf_url": "https://arxiv.org/pdf/2607.01812", + "primary_query": "llm-agent" + }, + { + "id": "2607.02453", + "title": "Adoption and Ecosystem Health: A Longitudinal Analysis of Open-Source Multi-Agent Frameworks", + "url": "https://arxiv.org/abs/2607.02453", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Xi Zhang", + "Papi Menon", + "Vivian Chu", + "Koray Cosguner" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.02453", + "source": "arxiv", + "source_id": "arxiv:2607.02453", + "pdf_url": "https://arxiv.org/pdf/2607.02453", + "primary_query": "ai-agent" + }, + { + "id": "2607.02245", + "title": "Copewell: A Multi-Agent Swarm Architecture for Equitable Mental Wellness Support", + "url": "https://arxiv.org/abs/2607.02245", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Seren Yenikent", + "Jack Vinijtrongjit", + "Katherine Ng" + ], + "categories": [ + "cs.AI", + "cs.CY", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.02245", + "source": "arxiv", + "source_id": "arxiv:2607.02245", + "pdf_url": "https://arxiv.org/pdf/2607.02245", + "primary_query": "ai-agent" + }, + { + "id": "2607.02381", + "title": "HULAT2 at MER-TRANS 2026: Governed Multi-Agent Simplification for Spanish Easy-to-Read Generation", + "url": "https://arxiv.org/abs/2607.02381", + "published": "2026-07-02", + "updated": "2026-07-02", + "authors": [ + "Lourdes Moreno", + "Paloma Martínez", + "Marco Antonio Sanchez-Escudero", + "Miguel Domínguez-Gómez" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2607.02381", + "source": "arxiv", + "source_id": "arxiv:2607.02381", + "pdf_url": "https://arxiv.org/pdf/2607.02381", + "primary_query": "agentic-ai" + }, + { + "id": "2607.01531", + "title": "OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration", + "url": "https://arxiv.org/abs/2607.01531", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "David Courtis", + "Wenhao Li", + "Scott Sanner" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent", + "planning-agent" + ], + "arxiv_id": "2607.01531", + "source": "arxiv", + "source_id": "arxiv:2607.01531", + "pdf_url": "https://arxiv.org/pdf/2607.01531", + "primary_query": "llm-agent" + }, + { + "id": "2607.01047", + "title": "Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates", + "url": "https://arxiv.org/abs/2607.01047", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Elias Najarro", + "Ane Espeseth", + "Eleni Nisioti", + "Sebastian Risi", + "Stefano Nichele" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "memory", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2607.01047", + "source": "arxiv", + "source_id": "arxiv:2607.01047", + "pdf_url": "https://arxiv.org/pdf/2607.01047", + "primary_query": "llm-agent" + }, + { + "id": "2607.01510", + "title": "Janus: a Playground for User-Involved Agentic Permission Management", + "url": "https://arxiv.org/abs/2607.01510", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Natalie Grace Brigham", + "Eugene Bagdasarian", + "Tadayoshi Kohno", + "Franziska Roesner" + ], + "categories": [ + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.01510", + "source": "arxiv", + "source_id": "arxiv:2607.01510", + "pdf_url": "https://arxiv.org/pdf/2607.01510", + "primary_query": "ai-agent" + }, + { + "id": "2607.00407", + "title": "Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising", + "url": "https://arxiv.org/abs/2607.00407", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Tianci Liu", + "Zihan Dong", + "Linjun Zhang", + "Haoyu Wang", + "jing Gao", + "Emre Kiciman", + "Ranveer Chandra", + "Wei-Ting Chen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "multi-agent", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2607.00407", + "source": "arxiv", + "source_id": "arxiv:2607.00407", + "pdf_url": "https://arxiv.org/pdf/2607.00407", + "primary_query": "ai-agent" + }, + { + "id": "2607.01366", + "title": "Auto-FL-Research: Agentic Search for Federated Learning Algorithms", + "url": "https://arxiv.org/abs/2607.01366", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Holger R. Roth", + "Ziyue Xu", + "Chester Chen", + "Daguang Xu", + "Peter Cnudde", + "Andrew Feng" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2607.01366", + "source": "arxiv", + "source_id": "arxiv:2607.01366", + "pdf_url": "https://arxiv.org/pdf/2607.01366", + "primary_query": "agentic-ai" + }, + { + "id": "2607.00990", + "title": "SWE-Doctor: Guiding Software Engineering Agents with Runtime Diagnosis from Multi-Faceted Bug Reproduction Tests", + "url": "https://arxiv.org/abs/2607.00990", + "published": "2026-07-01", + "updated": "2026-07-01", + "authors": [ + "Yaoqi Guo", + "Yang Liu", + "Jie M. Zhang", + "Yun Ma", + "Yiling Lou", + "Zhenpeng Chen" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2607.00990", + "source": "arxiv", + "source_id": "arxiv:2607.00990", + "pdf_url": "https://arxiv.org/pdf/2607.00990", + "primary_query": "coding-agent" + }, + { + "id": "2606.31471", + "title": "Think While You Map: Asynchronous Vision-Language Agents for Incremental 3D Scene Graphs", + "url": "https://arxiv.org/abs/2606.31471", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Deniz Bickici", + "Michael Pabst", + "Shohei Mori", + "Dieter Schmalstieg" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.31471", + "source": "arxiv", + "source_id": "arxiv:2606.31471", + "pdf_url": "https://arxiv.org/pdf/2606.31471", + "primary_query": "language-agent" + }, + { + "id": "2606.31831", + "title": "An Agentic AI Framework to Accelerate Scientific Discovery in Plant Phenotyping", + "url": "https://arxiv.org/abs/2606.31831", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Renan Souza", + "Daniel Rosendo", + "Kelsey Carter", + "John Lagergren", + "Frédéric Suter", + "Shelaine L. Curd", + "Gerald A. Tuskan", + "Rafael Ferreira da Silva", + "David Weston" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "ai-agent" + ], + "arxiv_id": "2606.31831", + "source": "arxiv", + "source_id": "arxiv:2606.31831", + "pdf_url": "https://arxiv.org/pdf/2606.31831", + "primary_query": "agentic-ai" + }, + { + "id": "2606.31916", + "title": "Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action", + "url": "https://arxiv.org/abs/2606.31916", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Ben Slater", + "Matteo G. Mecattaf", + "Lucy G. Cheke", + "John Burden", + "Winnie Street" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.31916", + "source": "arxiv", + "source_id": "arxiv:2606.31916", + "pdf_url": "https://arxiv.org/pdf/2606.31916", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.31767", + "title": "JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java", + "url": "https://arxiv.org/abs/2606.31767", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Khashayar Etemadi", + "Zhendong Su" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.31767", + "source": "arxiv", + "source_id": "arxiv:2606.31767", + "pdf_url": "https://arxiv.org/pdf/2606.31767", + "primary_query": "coding-agent" + }, + { + "id": "2606.31665", + "title": "ForecastAgentSearch: Towards a Multi-Expert Agent Search System for Geopolitical Event Forecasting", + "url": "https://arxiv.org/abs/2606.31665", + "published": "2026-06-30", + "updated": "2026-06-30", + "authors": [ + "Miaomiao Cai", + "He Chang", + "Yunshan Ma", + "See-kiong Ng" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.31665", + "source": "arxiv", + "source_id": "arxiv:2606.31665", + "pdf_url": "https://arxiv.org/pdf/2606.31665", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.29719", + "title": "A Diagnostic Framework and Multi-Evaluator Audit of Evaluator-Driven Preference Dynamics in Self-Adapting LLM Agents", + "url": "https://arxiv.org/abs/2606.29719", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Liu Zewen" + ], + "categories": [ + "cs.LG", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "llm-agent" + ], + "arxiv_id": "2606.29719", + "source": "arxiv", + "source_id": "arxiv:2606.29719", + "pdf_url": "https://arxiv.org/pdf/2606.29719", + "primary_query": "llm-agent" + }, + { + "id": "2606.29745", + "title": "ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit", + "url": "https://arxiv.org/abs/2606.29745", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Abhijnan Nath", + "Nikhil Krishnaswamy" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.29745", + "source": "arxiv", + "source_id": "arxiv:2606.29745", + "pdf_url": "https://arxiv.org/pdf/2606.29745", + "primary_query": "language-agent" + }, + { + "id": "2606.30970", + "title": "Behavioral Governance for Autonomous AI Agents: The AgentBound Framework", + "url": "https://arxiv.org/abs/2606.30970", + "published": "2026-06-29", + "updated": "2026-07-01", + "authors": [ + "Anuj Kaul", + "Qianlong Lan", + "Pranay Gupta" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.30970", + "source": "arxiv", + "source_id": "arxiv:2606.30970", + "pdf_url": "https://arxiv.org/pdf/2606.30970", + "primary_query": "ai-agent" + }, + { + "id": "2606.29788", + "title": "MemLeak: Diagnosing Information Leaks in Multimodal Agent Memory", + "url": "https://arxiv.org/abs/2606.29788", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Kuan Wang", + "Chao Zhang" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "memory" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory", + "ai-agent" + ], + "arxiv_id": "2606.29788", + "source": "arxiv", + "source_id": "arxiv:2606.29788", + "pdf_url": "https://arxiv.org/pdf/2606.29788", + "primary_query": "agent-memory" + }, + { + "id": "2606.30616", + "title": "Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent", + "url": "https://arxiv.org/abs/2606.30616", + "published": "2026-06-29", + "updated": "2026-06-29", + "authors": [ + "Lei Bai", + "Zongsheng Cao", + "Yang Chen", + "Zhiyao Cui", + "Shangheng Du", + "Yue Fan", + "Shiyang Feng", + "Zijie Guo", + "Haonan He", + "Liang He", + "Xiaohan He", + "Shuyue Hu", + "Yusong Hu", + "Songtao Huang", + "Yichen Jiang", + "Hao Li", + "Xin Li", + "Dahua Lin", + "Weihao Lin", + "Fenghua Ling", + "Dongrui Liu", + "Zhuo Liu", + "Runmin Ma", + "Chunjiang Mu", + "Haoyang Peng", + "Tianshuo Peng", + "Jinxin Shi", + "Luohe Shi", + "Boyuan Sun", + "Zelin Tan", + "Shengji Tang", + "Qianyi Wang", + "Yiming Wu", + "Yi Xie", + "Xiangchao Yan", + "Jingqi Ye", + "Peng Ye", + "Fangchen Yu", + "Jiakang Yuan", + "Bihao Zhan", + "Bo Zhang", + "Chen Zhang", + "Shufei Zhang", + "Shuaiyu Zhang", + "Wenlong Zhang", + "Yiqun Zhang", + "Junpeng Zhao", + "Zhijie Zhong", + "Bowen Zhou", + "Yuhao Zhou" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.30616", + "source": "arxiv", + "source_id": "arxiv:2606.30616", + "pdf_url": "https://arxiv.org/pdf/2606.30616", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.30111", + "title": "Automating the Design of Embodied Agent Architectures", + "url": "https://arxiv.org/abs/2606.30111", + "published": "2026-06-29", + "updated": "2026-07-03", + "authors": [ + "Jian Zhou", + "Sihao Lin", + "Jin Li", + "Shuai Fu", + "Gengze Zhou", + "Qi Wu" + ], + "categories": [ + "cs.RO", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.30111", + "source": "arxiv", + "source_id": "arxiv:2606.30111", + "pdf_url": "https://arxiv.org/pdf/2606.30111", + "primary_query": "coding-agent" + }, + { + "id": "2606.29495", + "title": "Cognitive World Models for Process-Level Social Influence Evaluation", + "url": "https://arxiv.org/abs/2606.29495", + "published": "2026-06-28", + "updated": "2026-06-28", + "authors": [ + "Minghui Ma", + "Bin Guo", + "Han Wang", + "Mengqi Chen", + "Jingqi Liu", + "Yan Liu", + "Zhiwen Yu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.29495", + "source": "arxiv", + "source_id": "arxiv:2606.29495", + "pdf_url": "https://arxiv.org/pdf/2606.29495", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.28896", + "title": "A Task-Driven and Quality-Assured Agent Framework for SAR Data Generation", + "url": "https://arxiv.org/abs/2606.28896", + "published": "2026-06-27", + "updated": "2026-06-27", + "authors": [ + "Xuanting Wu", + "Fan Zhanga", + "Fei Ma", + "Ling Guan", + "Guochun Ma", + "Yongsheng Zhou" + ], + "categories": [ + "eess.IV", + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.28896", + "source": "arxiv", + "source_id": "arxiv:2606.28896", + "pdf_url": "https://arxiv.org/pdf/2606.28896", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.28279", + "title": "Agentic Hardware Design as Repository-Level Code Evolution", + "url": "https://arxiv.org/abs/2606.28279", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Cunxi Yu", + "Chenhui Deng", + "Nathaniel Pinckney", + "Brucek Khailany" + ], + "categories": [ + "cs.AR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.28279", + "source": "arxiv", + "source_id": "arxiv:2606.28279", + "pdf_url": "https://arxiv.org/pdf/2606.28279", + "primary_query": "agentic-ai" + }, + { + "id": "2606.28430", + "title": "Building to the Test: Coding Agents Deliver What You Check, Not What You Requested", + "url": "https://arxiv.org/abs/2606.28430", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Yanuo Ma", + "Ben Kereopa-Yorke", + "Ben Schultz" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.28430", + "source": "arxiv", + "source_id": "arxiv:2606.28430", + "pdf_url": "https://arxiv.org/pdf/2606.28430", + "primary_query": "coding-agent" + }, + { + "id": "2606.28187", + "title": "GBC: Gradient-Based Connections for Optimizing Multi-Agent Systems", + "url": "https://arxiv.org/abs/2606.28187", + "published": "2026-06-26", + "updated": "2026-06-26", + "authors": [ + "Xiaocheng Yang", + "Abdulrahman Alrabah", + "Dilek Hakkani-Tür", + "Gokhan Tur" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "multi-agent", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "multi-agent-llm" + ], + "arxiv_id": "2606.28187", + "source": "arxiv", + "source_id": "arxiv:2606.28187", + "pdf_url": "https://arxiv.org/pdf/2606.28187", + "primary_query": "multi-agent-llm" + }, + { + "id": "2606.27416", + "title": "Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents", + "url": "https://arxiv.org/abs/2606.27416", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Vassili Philippov", + "Pavel Katunin", + "Dmitry Andreev", + "Igor Ostanin", + "Anton Nikolaev" + ], + "categories": [ + "cs.MA", + "cs.SE" + ], + "topics": [ + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.27416", + "source": "arxiv", + "source_id": "arxiv:2606.27416", + "pdf_url": "https://arxiv.org/pdf/2606.27416", + "primary_query": "coding-agent" + }, + { + "id": "2606.26649", + "title": "Autoformalization of Agent Instructions into Policy-as-Code", + "url": "https://arxiv.org/abs/2606.26649", + "published": "2026-06-25", + "updated": "2026-06-25", + "authors": [ + "Adam Mondl", + "Matthew Maisel", + "John H. Brock" + ], + "categories": [ + "cs.AI", + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.26649", + "source": "arxiv", + "source_id": "arxiv:2606.26649", + "pdf_url": "https://arxiv.org/pdf/2606.26649", + "primary_query": "agent-safety" + }, + { + "id": "2606.26057", + "title": "The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems", + "url": "https://arxiv.org/abs/2606.26057", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Seth Dobrin", + "Łukasz Chmiel" + ], + "categories": [ + "cs.AI", + "cs.CR", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.26057", + "source": "arxiv", + "source_id": "arxiv:2606.26057", + "pdf_url": "https://arxiv.org/pdf/2606.26057", + "primary_query": "ai-agent" + }, + { + "id": "2606.26356", + "title": "Instruction Bleed: Cross-Module Interference in Prompt-Composed Agentic Systems", + "url": "https://arxiv.org/abs/2606.26356", + "published": "2026-06-24", + "updated": "2026-06-24", + "authors": [ + "Ching-Yu Lin", + "Yifan Liu" + ], + "categories": [ + "cs.AI", + "cs.IR", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "agentic-ai" + ], + "arxiv_id": "2606.26356", + "source": "arxiv", + "source_id": "arxiv:2606.26356", + "pdf_url": "https://arxiv.org/pdf/2606.26356", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.25519", + "title": "Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models", + "url": "https://arxiv.org/abs/2606.25519", + "published": "2026-06-24", + "updated": "2026-06-29", + "authors": [ + "Xinyu Lian", + "Walid Krichene", + "Beichen Huang", + "Masahiro Tanaka", + "Olatunji Ruwase", + "Li Zhang", + "Minjia Zhang" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.25519", + "source": "arxiv", + "source_id": "arxiv:2606.25519", + "pdf_url": "https://arxiv.org/pdf/2606.25519", + "primary_query": "tool-use" + }, + { + "id": "2606.25195", + "title": "SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward", + "url": "https://arxiv.org/abs/2606.25195", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Rupam Patir", + "Keyan Guo", + "Haipeng Cai", + "Hongxin Hu" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "coding-agent" + ], + "arxiv_id": "2606.25195", + "source": "arxiv", + "source_id": "arxiv:2606.25195", + "pdf_url": "https://arxiv.org/pdf/2606.25195", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24416", + "title": "Agentic AI for Bilevel Long-Term Optimization of Policy-Driven Physical Layer Systems", + "url": "https://arxiv.org/abs/2606.24416", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Bingnan Xiao", + "Chenhao Yang", + "Wei Ni", + "Xin Wang", + "Tony Q. S. Quek" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.24416", + "source": "arxiv", + "source_id": "arxiv:2606.24416", + "pdf_url": "https://arxiv.org/pdf/2606.24416", + "primary_query": "agentic-ai" + }, + { + "id": "2606.24855", + "title": "OpenThoughts-Agent: Data Recipes for Agentic Models", + "url": "https://arxiv.org/abs/2606.24855", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Negin Raoof", + "Richard Zhuang", + "Marianna Nezhurina", + "Etash Guha", + "Atula Tejaswi", + "Ryan Marten", + "Charlie F. Ruan", + "Tyler Griggs", + "Alexander Glenn Shaw", + "Hritik Bansal", + "E. Kelly Buchanan", + "Artem Gazizov", + "Reinhard Heckel", + "Chinmay Hegde", + "Sankalp Jajee", + "Daanish Khazi", + "Emmanouil Koukoumidis", + "Xiangyi Li", + "Hange Liu", + "Shlok Natarajan", + "Harsh Raj", + "Nicholas Roberts", + "Ethan Shen", + "Nishad Singhi", + "Michael Siu", + "Ashima Suvarna", + "Hanwen Xing", + "Patrick Yubeaton", + "Robert Zhang", + "Leon Liangyu Chen", + "Xiaokun Chen", + "Steven Dillmann", + "Saadia Gabriel", + "Xunyi Jiang", + "Anurag Kashyap", + "Boxuan Li", + "Yein Park", + "Minh Pham", + "Sujay Sanghavi", + "Lin Shi", + "Ke Sun", + "Yixin Wang", + "Zhiwei Xu", + "Erica Zhang", + "Siyan Zhao", + "Wanjia Zhao", + "Jenia Jitsev", + "Alex Dimakis", + "Benjamin Feuer", + "Ludwig Schmidt" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.24855", + "source": "arxiv", + "source_id": "arxiv:2606.24855", + "pdf_url": "https://arxiv.org/pdf/2606.24855", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.25191", + "title": "To Isolate or to Score? Model-Adaptive Assessment for Cost-Efficient Multi-Agent RAG", + "url": "https://arxiv.org/abs/2606.25191", + "published": "2026-06-23", + "updated": "2026-06-23", + "authors": [ + "Jungseob Lee", + "Chanjun Park", + "Heuiseok Lim" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.25191", + "source": "arxiv", + "source_id": "arxiv:2606.25191", + "pdf_url": "https://arxiv.org/pdf/2606.25191", + "primary_query": "rag-agent" + }, + { + "id": "2606.23130", + "title": "Understanding the (In)Security of Vibe-Coded Applications", + "url": "https://arxiv.org/abs/2606.23130", + "published": "2026-06-22", + "updated": "2026-06-23", + "authors": [ + "Junquan Deng", + "Zhiyu Fan", + "Ruijie Meng" + ], + "categories": [ + "cs.CR", + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.23130", + "source": "arxiv", + "source_id": "arxiv:2606.23130", + "pdf_url": "https://arxiv.org/pdf/2606.23130", + "primary_query": "ai-agent" + }, + { + "id": "2606.22953", + "title": "Plans Don't Persist: Why Context Management Is Load Bearing for LLM Agents", + "url": "https://arxiv.org/abs/2606.22953", + "published": "2026-06-22", + "updated": "2026-06-22", + "authors": [ + "Aman Mehta", + "Anupam Datta" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.22953", + "source": "arxiv", + "source_id": "arxiv:2606.22953", + "pdf_url": "https://arxiv.org/pdf/2606.22953", + "primary_query": "planning-agent" + }, + { + "id": "2606.21836", + "title": "AgentDSE: Reasoning-Augmented Architectural Design Space Exploration", + "url": "https://arxiv.org/abs/2606.21836", + "published": "2026-06-20", + "updated": "2026-06-20", + "authors": [ + "Chenyu Wang", + "Jiahe Caroline Shi", + "David Kong", + "Duane S. Boning", + "Zishen Wan", + "Yilun Du", + "Vijay Janapa Reddi" + ], + "categories": [ + "cs.AR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.21836", + "source": "arxiv", + "source_id": "arxiv:2606.21836", + "pdf_url": "https://arxiv.org/pdf/2606.21836", + "primary_query": "coding-agent" + }, + { + "id": "2606.21401", + "title": "SwarmX: Agentic Scheduling for Low-Latency Agentic Systems", + "url": "https://arxiv.org/abs/2606.21401", + "published": "2026-06-19", + "updated": "2026-06-28", + "authors": [ + "Yeqi Huang", + "Yanwei Ye", + "Guomin Chen", + "Wenhao Su", + "Bin Gong", + "Jialian Li", + "Zhan Lu", + "Yangshen Deng", + "Xuan Sun", + "Le Xu", + "Luo Mai" + ], + "categories": [ + "cs.DC", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai" + ], + "arxiv_id": "2606.21401", + "source": "arxiv", + "source_id": "arxiv:2606.21401", + "pdf_url": "https://arxiv.org/pdf/2606.21401", + "primary_query": "agentic-ai" + }, + { + "id": "2606.21228", + "title": "Sakana Fugu Technical Report", + "url": "https://arxiv.org/abs/2606.21228", + "published": "2026-06-19", + "updated": "2026-06-23", + "authors": [ + "Yujin Tang", + "Edoardo Cetin", + "Jinglue Xu", + "Qi Sun", + "Stefan Nielsen", + "Vincent Richard", + "Haruto Goda", + "Iaroslav Tymchenko", + "Nhan Nguyen", + "Hyunin Lee", + "Mari Ashiga", + "Shashank Kotyan", + "So Kuroki", + "Tarin Clanuwat" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "coding-agent", + "multi-agent", + "rag", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "coding-agent" + ], + "arxiv_id": "2606.21228", + "source": "arxiv", + "source_id": "arxiv:2606.21228", + "pdf_url": "https://arxiv.org/pdf/2606.21228", + "primary_query": "coding-agent" + }, + { + "id": "2606.20510", + "title": "Efficient and Sound Probabilistic Verification for AI Agents", + "url": "https://arxiv.org/abs/2606.20510", + "published": "2026-06-18", + "updated": "2026-06-18", + "authors": [ + "Alaia Solko-Breslin", + "Pramod Kaushik Mudrakarta", + "Mihai Christodorescu", + "Somesh Jha", + "Krishnamurthy Dj Dvijotham" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent" + ], + "arxiv_id": "2606.20510", + "source": "arxiv", + "source_id": "arxiv:2606.20510", + "pdf_url": "https://arxiv.org/pdf/2606.20510", + "primary_query": "ai-agent" + }, + { + "id": "2606.19242", + "title": "Runtime Compliance Verification for AI Agents", + "url": "https://arxiv.org/abs/2606.19242", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Nafiseh Kahani", + "Masoud Barati", + "Diana Addae" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "ai-agent", + "function-calling", + "tool-use" + ], + "arxiv_id": "2606.19242", + "source": "arxiv", + "source_id": "arxiv:2606.19242", + "pdf_url": "https://arxiv.org/pdf/2606.19242", + "primary_query": "ai-agent" + }, + { + "id": "2606.28374", + "title": "Recursive Self-Evolving Agents via Held-Out Selection", + "url": "https://arxiv.org/abs/2606.28374", + "published": "2026-06-17", + "updated": "2026-06-17", + "authors": [ + "Michael Nguyen", + "Quoc Nguyen", + "Paul Vuong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.28374", + "source": "arxiv", + "source_id": "arxiv:2606.28374", + "pdf_url": "https://arxiv.org/pdf/2606.28374", + "primary_query": "tool-use" + }, + { + "id": "2606.18363", + "title": "Guava: An Effective and Universal Harness for Embodied Manipulation", + "url": "https://arxiv.org/abs/2606.18363", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Haowen Liu", + "Xirui Li", + "Shaoxiong Yao", + "Peng Shi", + "Tianyi Zhou", + "Jia-Bin Huang", + "Furong Huang", + "Jiayuan Mao" + ], + "categories": [ + "cs.RO", + "cs.AI" + ], + "topics": [ + "embodied-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agentic-ai", + "tool-use" + ], + "arxiv_id": "2606.18363", + "source": "arxiv", + "source_id": "arxiv:2606.18363", + "pdf_url": "https://arxiv.org/pdf/2606.18363", + "primary_query": "agentic-ai" + }, + { + "id": "2606.17453", + "title": "MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors", + "url": "https://arxiv.org/abs/2606.17453", + "published": "2026-06-16", + "updated": "2026-06-17", + "authors": [ + "Lubin Bai", + "Mengyu Cao", + "Sixue Wang", + "Zhongwei Wan", + "Yue Pan", + "Jiale Hou", + "Xiang Li", + "Xiuyuan Zhang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.17453", + "source": "arxiv", + "source_id": "arxiv:2606.17453", + "pdf_url": "https://arxiv.org/pdf/2606.17453", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18023", + "title": "LoopCoder-v2: Only Loop Once for Efficient Test-Time Computation Scaling", + "url": "https://arxiv.org/abs/2606.18023", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Jian Yang", + "Shawn Guo", + "Wei Zhang", + "Tianyu Zheng", + "Yaxin Du", + "Haau-Sing Li", + "Jiajun Wu", + "Yue Song", + "Yan Xing", + "Qingsong Cai", + "Zelong Huang", + "Chuan Hao", + "Ran Tao", + "Xianglong Liu", + "Wayne Xin Zhao", + "Mingjie Tang", + "Weifeng Lv", + "Ming Zhou", + "Bryan Dai" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.18023", + "source": "arxiv", + "source_id": "arxiv:2606.18023", + "pdf_url": "https://arxiv.org/pdf/2606.18023", + "primary_query": "tool-use" + }, + { + "id": "2606.17383", + "title": "Model Validation of Agentic AI Systems: A POMDP-Based Framework for Belief-State, Forecast, and Policy Validation", + "url": "https://arxiv.org/abs/2606.17383", + "published": "2026-06-16", + "updated": "2026-06-16", + "authors": [ + "Matthew Francis Dixon" + ], + "categories": [ + "q-fin.RM", + "cs.AI", + "cs.LG", + "stat.ML" + ], + "topics": [ + "agent-safety", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.17383", + "source": "arxiv", + "source_id": "arxiv:2606.17383", + "pdf_url": "https://arxiv.org/pdf/2606.17383", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.16813", + "title": "GIST-CMTF: Goal-State Inference for Causal Minimal Tool Filtering in LLM Agents", + "url": "https://arxiv.org/abs/2606.16813", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Rahul Suresh Babu", + "Rohit Shukla" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.16813", + "source": "arxiv", + "source_id": "arxiv:2606.16813", + "pdf_url": "https://arxiv.org/pdf/2606.16813", + "primary_query": "tool-use" + }, + { + "id": "2606.16839", + "title": "Towards LLM Accelerated Rapid Reviews for Software Tool Discovery -- Case for Log Anomaly Detection", + "url": "https://arxiv.org/abs/2606.16839", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Jesse Nyyssölä", + "Hamza Bin Mazhar", + "Alexander Bakhtin", + "Matteo Esposito", + "Nana Reinikainen", + "Yuqing Wang", + "Ying Song", + "Davide Taibi", + "Mika Mäntylä" + ], + "categories": [ + "cs.SE" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.16839", + "source": "arxiv", + "source_id": "arxiv:2606.16839", + "pdf_url": "https://arxiv.org/pdf/2606.16839", + "primary_query": "planning-agent" + }, + { + "id": "2606.16481", + "title": "Steering Emotional Dynamics for Art Therapy: Controllable Narrative Script Generation through Hierarchically Guided LLM Agents", + "url": "https://arxiv.org/abs/2606.16481", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Suqing Wang", + "Qinghai Miao", + "Chao Guo", + "Yisheng Lv" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "computer-use", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.16481", + "source": "arxiv", + "source_id": "arxiv:2606.16481", + "pdf_url": "https://arxiv.org/pdf/2606.16481", + "primary_query": "planning-agent" + }, + { + "id": "2606.17368", + "title": "Distributed General-Purpose Agent Networks: Architecture, Key Mechanisms, and Prototypes", + "url": "https://arxiv.org/abs/2606.17368", + "published": "2026-06-15", + "updated": "2026-06-15", + "authors": [ + "Shengli Zhang", + "Deen Ma", + "Zibin Lin", + "Taotao Wang" + ], + "categories": [ + "cs.AI", + "cs.NI" + ], + "topics": [ + "computer-use", + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.17368", + "source": "arxiv", + "source_id": "arxiv:2606.17368", + "pdf_url": "https://arxiv.org/pdf/2606.17368", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.15709", + "title": "AI-Driven Framework for Adaptive Water Network Management with Proof-of-Concept Implementation: Addressing Non-Revenue Water in Jordan", + "url": "https://arxiv.org/abs/2606.15709", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Mohammed Fasha", + "Nahel Al-Maayta", + "Bilal Sowan", + "Mohammad Athamneh", + "Husam Barham" + ], + "categories": [ + "cs.AI", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling", + "rag-agent" + ], + "arxiv_id": "2606.15709", + "source": "arxiv", + "source_id": "arxiv:2606.15709", + "pdf_url": "https://arxiv.org/pdf/2606.15709", + "primary_query": "function-calling" + }, + { + "id": "2606.15874", + "title": "LLM-as-Code: Agentic Programming for Agent Harness", + "url": "https://arxiv.org/abs/2606.15874", + "published": "2026-06-14", + "updated": "2026-06-22", + "authors": [ + "Junjia Qi", + "Zichuan Fu", + "Jingtong Gao", + "Wenlin Zhang", + "Hanyu Yan", + "Xian Wu", + "Xiangyu Zhao" + ], + "categories": [ + "cs.AI", + "cs.SE" + ], + "topics": [ + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.15874", + "source": "arxiv", + "source_id": "arxiv:2606.15874", + "pdf_url": "https://arxiv.org/pdf/2606.15874", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.15906", + "title": "MAGE-RAG: Multigranular Adaptive Graph Evidence for Agentic Multimodal RAG in Long-Document QA", + "url": "https://arxiv.org/abs/2606.15906", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Yilong Zuo", + "Xunkai Li", + "Jing Yuan", + "Qiangqiang Dai", + "Hongchao Qin", + "Ronghua Li" + ], + "categories": [ + "cs.IR", + "cs.AI", + "cs.CL", + "cs.DB", + "cs.MM" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.15906", + "source": "arxiv", + "source_id": "arxiv:2606.15906", + "pdf_url": "https://arxiv.org/pdf/2606.15906", + "primary_query": "rag-agent" + }, + { + "id": "2606.15994", + "title": "Agentic Framework for Deep Learning workload migration via In-Context Learning", + "url": "https://arxiv.org/abs/2606.15994", + "published": "2026-06-14", + "updated": "2026-06-14", + "authors": [ + "Qiyue Liang", + "Steven Ingram", + "George Vanica", + "Andi Gavrilescu", + "Newfel Harrat", + "Hassan Sipra", + "Sethuraman Sankaran" + ], + "categories": [ + "cs.AI", + "cs.LG" + ], + "topics": [ + "agent-safety", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.15994", + "source": "arxiv", + "source_id": "arxiv:2606.15994", + "pdf_url": "https://arxiv.org/pdf/2606.15994", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.15034", + "title": "OSGuard: A Benchmark for Safety in Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.15034", + "published": "2026-06-13", + "updated": "2026-06-13", + "authors": [ + "Mina Mohammadmirzaei", + "Jeffrey Flanigan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.15034", + "source": "arxiv", + "source_id": "arxiv:2606.15034", + "pdf_url": "https://arxiv.org/pdf/2606.15034", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.12837", + "title": "LoHoSearch: Benchmarking Long-Horizon Search Agents Beyond the Human Difficulty Ceiling", + "url": "https://arxiv.org/abs/2606.12837", + "published": "2026-06-11", + "updated": "2026-06-17", + "authors": [ + "Jiarui Zhao", + "Rongzhi Zhang", + "Lingchuan Liu", + "Hao Yang", + "Xunliang Cai", + "Xi Su" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.12837", + "source": "arxiv", + "source_id": "arxiv:2606.12837", + "pdf_url": "https://arxiv.org/pdf/2606.12837", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.13663", + "title": "HyperTool: Beyond Step-Wise Tool Calls for Tool-Augmented Agents", + "url": "https://arxiv.org/abs/2606.13663", + "published": "2026-06-11", + "updated": "2026-06-11", + "authors": [ + "Yaxin Du", + "Yifan Zhou", + "Yujie Ge", + "Jiajun Wang", + "Xianghe Pang", + "Shuo Tang", + "Tuney Zheng", + "Bryan Dai", + "Jian Yang", + "Siheng Chen" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.13663", + "source": "arxiv", + "source_id": "arxiv:2606.13663", + "pdf_url": "https://arxiv.org/pdf/2606.13663", + "primary_query": "tool-use" + }, + { + "id": "2606.12634", + "title": "Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents", + "url": "https://arxiv.org/abs/2606.12634", + "published": "2026-06-10", + "updated": "2026-06-29", + "authors": [ + "Tianyu Ding", + "Jianhong Xin", + "Juan Pablo De la Cruz Weinstein" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "tool-use" + ], + "arxiv_id": "2606.12634", + "source": "arxiv", + "source_id": "arxiv:2606.12634", + "pdf_url": "https://arxiv.org/pdf/2606.12634", + "primary_query": "tool-use" + }, + { + "id": "2606.11688", + "title": "Goal-Autopilot: A Verifiable Anti-Fabrication Firewall for Unattended Long-Horizon Agents", + "url": "https://arxiv.org/abs/2606.11688", + "published": "2026-06-10", + "updated": "2026-06-10", + "authors": [ + "Youwang Deng" + ], + "categories": [ + "cs.CL", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.11688", + "source": "arxiv", + "source_id": "arxiv:2606.11688", + "pdf_url": "https://arxiv.org/pdf/2606.11688", + "primary_query": "planning-agent" + }, + { + "id": "2606.10394", + "title": "STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios", + "url": "https://arxiv.org/abs/2606.10394", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Sirui Liang", + "Bohan Yu", + "Peiyu Wang", + "Shiguang Guo", + "Wenxing Hu", + "Pengfei Cao", + "Jian Zhao", + "Cao Liu", + "Ke Zeng", + "Xunliang Cai", + "Kang Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.10394", + "source": "arxiv", + "source_id": "arxiv:2606.10394", + "pdf_url": "https://arxiv.org/pdf/2606.10394", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.10532", + "title": "ActiveMem: Distributed Active Memory for Long-Horizon LLM Reasoning", + "url": "https://arxiv.org/abs/2606.10532", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Yunhan Jiang", + "Wenbin Duan", + "Shasha Guo", + "Liang Pang", + "Xiaoqian Sun", + "Huawei Shen" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.10532", + "source": "arxiv", + "source_id": "arxiv:2606.10532", + "pdf_url": "https://arxiv.org/pdf/2606.10532", + "primary_query": "agent-memory" + }, + { + "id": "2606.11176", + "title": "Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories", + "url": "https://arxiv.org/abs/2606.11176", + "published": "2026-06-09", + "updated": "2026-06-09", + "authors": [ + "Kevin Qinghong Lin", + "Batu EI", + "Yuhong Shi", + "Pan Lu", + "Philip Torr", + "James Zou" + ], + "categories": [ + "cs.CV", + "cs.CL", + "cs.CY", + "cs.HC" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.11176", + "source": "arxiv", + "source_id": "arxiv:2606.11176", + "pdf_url": "https://arxiv.org/pdf/2606.11176", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.08960", + "title": "Hardening Agent Benchmarks with Adversarial Hacker-Fixer Loops", + "url": "https://arxiv.org/abs/2606.08960", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Ziqian Zhong", + "Ivgeni Segal", + "Ivan Bercovich", + "Shashwat Saxena", + "Kexun Zhang", + "Aditi Raghunathan" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.LG", + "cs.MA" + ], + "topics": [ + "agent-evaluation", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.08960", + "source": "arxiv", + "source_id": "arxiv:2606.08960", + "pdf_url": "https://arxiv.org/pdf/2606.08960", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.09447", + "title": "AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning", + "url": "https://arxiv.org/abs/2606.09447", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Bojie Rong", + "Zheyu Shen", + "Qiaoping Wang", + "Pengfei Kang", + "Yang Xu", + "Yawen Wei", + "Hanyu Wu", + "Zhi Zhao", + "Leihao Pei", + "Linquan Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.09447", + "source": "arxiv", + "source_id": "arxiv:2606.09447", + "pdf_url": "https://arxiv.org/pdf/2606.09447", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09426", + "title": "WeaveBench: A Long-Horizon, Real-World Benchmark for Computer-Use Agents with Hybrid Interfaces", + "url": "https://arxiv.org/abs/2606.09426", + "published": "2026-06-08", + "updated": "2026-07-06", + "authors": [ + "Wanli Li", + "Bowen Zhou", + "Yunyao Yu", + "Zhou Xu", + "Yifan Yang", + "Dongsheng Li", + "Caihua Shan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "web-gui-agent" + ], + "arxiv_id": "2606.09426", + "source": "arxiv", + "source_id": "arxiv:2606.09426", + "pdf_url": "https://arxiv.org/pdf/2606.09426", + "primary_query": "web-gui-agent" + }, + { + "id": "2606.09316", + "title": "Anything2Skill: Compiling External Knowledge into Reusable Skills for Agents", + "url": "https://arxiv.org/abs/2606.09316", + "published": "2026-06-08", + "updated": "2026-06-19", + "authors": [ + "Qianjun Pan", + "Yutao Yang", + "Junsong Li", + "Jie Zhou", + "Kai Chen", + "Xin Li", + "Qin Chen", + "Liang He" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.09316", + "source": "arxiv", + "source_id": "arxiv:2606.09316", + "pdf_url": "https://arxiv.org/pdf/2606.09316", + "primary_query": "rag-agent" + }, + { + "id": "2606.09961", + "title": "3SPO: State-Score-Supervised Policy Optimization for LLM Agents", + "url": "https://arxiv.org/abs/2606.09961", + "published": "2026-06-08", + "updated": "2026-06-08", + "authors": [ + "Yu Han", + "Kailing Li", + "Yang Jiao", + "Yulin Dai", + "Yuqian Fu", + "Linhai Zhuo", + "Tianwen Qian" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "computer-use", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.09961", + "source": "arxiv", + "source_id": "arxiv:2606.09961", + "pdf_url": "https://arxiv.org/pdf/2606.09961", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.08172", + "title": "The Governance of Human-LLM Interaction: Safety Gating, Civility Steering, and Affective Default Lock-In", + "url": "https://arxiv.org/abs/2606.08172", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Manuele Reani", + "Hongjian Zhang", + "Hongyu Tian" + ], + "categories": [ + "cs.HC", + "cs.AI", + "cs.CY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.08172", + "source": "arxiv", + "source_id": "arxiv:2606.08172", + "pdf_url": "https://arxiv.org/pdf/2606.08172", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.08162", + "title": "Silent Failure in LLM Agent Systems: The Entropy Principle and the Inevitable Disorder of Autonomous Agents", + "url": "https://arxiv.org/abs/2606.08162", + "published": "2026-06-06", + "updated": "2026-06-06", + "authors": [ + "Dexing Liu" + ], + "categories": [ + "cs.MA" + ], + "topics": [ + "memory", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.08162", + "source": "arxiv", + "source_id": "arxiv:2606.08162", + "pdf_url": "https://arxiv.org/pdf/2606.08162", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.07836", + "title": "Agentic multi-fidelity learning of quasiparticle and excitonic properties", + "url": "https://arxiv.org/abs/2606.07836", + "published": "2026-06-05", + "updated": "2026-06-05", + "authors": [ + "Arnab Neogi", + "Aaron Forde", + "Christopher A. Lane", + "Sergei Tretiak", + "Jian-Xin Zhu" + ], + "categories": [ + "cond-mat.mtrl-sci", + "cond-mat.stat-mech", + "cs.AI", + "physics.comp-ph", + "quant-ph" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "workflow-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.07836", + "source": "arxiv", + "source_id": "arxiv:2606.07836", + "pdf_url": "https://arxiv.org/pdf/2606.07836", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.18272", + "title": "Mitigating Anchoring Bias in LLM-Based Agents for Energy-Efficient 6G Autonomous Networks", + "url": "https://arxiv.org/abs/2606.18272", + "published": "2026-06-05", + "updated": "2026-06-18", + "authors": [ + "Hatim Chergui", + "Claudia Carballo González", + "Farhad Rezazadeh", + "Merouane Debbah" + ], + "categories": [ + "cs.NI", + "cs.AI", + "eess.SY" + ], + "topics": [ + "agent-safety", + "computer-use", + "multi-agent", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.18272", + "source": "arxiv", + "source_id": "arxiv:2606.18272", + "pdf_url": "https://arxiv.org/pdf/2606.18272", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2606.05658", + "title": "Agent-Orchestrated Adaptive RAG: A Comparative Study on Structured and Multi-Hop Retrieval", + "url": "https://arxiv.org/abs/2606.05658", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Anuj Maharjan", + "Devinder Kaur", + "Richard Molyet" + ], + "categories": [ + "cs.IR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.05658", + "source": "arxiv", + "source_id": "arxiv:2606.05658", + "pdf_url": "https://arxiv.org/pdf/2606.05658", + "primary_query": "rag-agent" + }, + { + "id": "2606.05622", + "title": "AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints", + "url": "https://arxiv.org/abs/2606.05622", + "published": "2026-06-04", + "updated": "2026-06-04", + "authors": [ + "Jiayu Liu", + "Cheng Qian", + "Zhenhailong Wang", + "Bingxuan Li", + "Jiateng Liu", + "Heng Wang", + "Jeonghwan Kim", + "Yumeng Wang", + "Xiusi Chen", + "Yi R. Fung", + "Heng Ji" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2606.05622", + "source": "arxiv", + "source_id": "arxiv:2606.05622", + "pdf_url": "https://arxiv.org/pdf/2606.05622", + "primary_query": "planning-agent" + }, + { + "id": "2606.05436", + "title": "Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison", + "url": "https://arxiv.org/abs/2606.05436", + "published": "2026-06-03", + "updated": "2026-06-03", + "authors": [ + "Alejandro Lozano", + "Keiko Ihara", + "Ping-Hao Yang", + "Carrie E. Robertson", + "Jennifer Stern", + "Allan Purdy", + "Hsiangkuo Yuan", + "Pengfei Zhang", + "Yulia Orlova", + "Olga Fermo", + "Jennifer Hranilovich", + "Fred Cohen", + "Todd J. Schwedt", + "Jenelle A. Jindal", + "Serena Yeung-Levy", + "Chia-Chun Chiang" + ], + "categories": [ + "cs.AI", + "cs.CL", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.05436", + "source": "arxiv", + "source_id": "arxiv:2606.05436", + "pdf_url": "https://arxiv.org/pdf/2606.05436", + "primary_query": "rag-agent" + }, + { + "id": "2606.03544", + "title": "SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems", + "url": "https://arxiv.org/abs/2606.03544", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Linyue Pan", + "Yaoming Zhu", + "Lin Qiu", + "Xuezhi Cao", + "Xunliang Cai" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2606.03544", + "source": "arxiv", + "source_id": "arxiv:2606.03544", + "pdf_url": "https://arxiv.org/pdf/2606.03544", + "primary_query": "language-agent" + }, + { + "id": "2606.03157", + "title": "ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models", + "url": "https://arxiv.org/abs/2606.03157", + "published": "2026-06-02", + "updated": "2026-06-02", + "authors": [ + "Ruihui Hou", + "Siyi Zhu", + "Ziyue Huai", + "Guangya Yu", + "Yongqi Fan", + "Chunming Wang", + "Tong Ruan" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.03157", + "source": "arxiv", + "source_id": "arxiv:2606.03157", + "pdf_url": "https://arxiv.org/pdf/2606.03157", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.01961", + "title": "AutoMedBench: Towards Medical AutoResearch with Agentic AI Models", + "url": "https://arxiv.org/abs/2606.01961", + "published": "2026-06-01", + "updated": "2026-06-03", + "authors": [ + "Junqi Liu", + "Selena Song", + "Yuhan Wang", + "Jiawei Mao", + "Hardy Chen", + "Xiaoke Huang", + "Tianhao Qi", + "Pengfei Guo", + "Yucheng Tang", + "Yufan He", + "Can Zhao", + "Andriy Myronenko", + "Dong Yang", + "Daguang Xu", + "Yuyin Zhou" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.01961", + "source": "arxiv", + "source_id": "arxiv:2606.01961", + "pdf_url": "https://arxiv.org/pdf/2606.01961", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.01185", + "title": "\"Skill issues'': data-centric optimization of lakehouse agents", + "url": "https://arxiv.org/abs/2606.01185", + "published": "2026-05-31", + "updated": "2026-05-31", + "authors": [ + "Nicole Rose Schneider", + "Davide Ghilardi", + "Giacomo Piccinini", + "Jacopo Tagliabue" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2606.01185", + "source": "arxiv", + "source_id": "arxiv:2606.01185", + "pdf_url": "https://arxiv.org/pdf/2606.01185", + "primary_query": "agent-evaluation" + }, + { + "id": "2606.01138", + "title": "memorywire: A Vendor-Neutral Wire Format for Agent Memory Operations", + "url": "https://arxiv.org/abs/2606.01138", + "published": "2026-05-31", + "updated": "2026-06-03", + "authors": [ + "Thamilvendhan Munirathinam" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.DC" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2606.01138", + "source": "arxiv", + "source_id": "arxiv:2606.01138", + "pdf_url": "https://arxiv.org/pdf/2606.01138", + "primary_query": "agent-memory" + }, + { + "id": "2606.01166", + "title": "BraveGuard: From Open-World Threats to Safer Computer-Use Agents", + "url": "https://arxiv.org/abs/2606.01166", + "published": "2026-05-31", + "updated": "2026-06-02", + "authors": [ + "Yunhao Feng", + "Xiaohu Du", + "Xinhao Deng", + "Yifan Ding", + "Ming Wen", + "Yixu Wang", + "Yuxiang Xie", + "Baihui Zheng", + "Yingshui Tan", + "Yige Li", + "Yutao Wu", + "Kerui Cao", + "Wenke Huang", + "Yanming Guo", + "Xingjun Ma", + "Yu-Gang Jiang" + ], + "categories": [ + "cs.CR", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2606.01166", + "source": "arxiv", + "source_id": "arxiv:2606.01166", + "pdf_url": "https://arxiv.org/pdf/2606.01166", + "primary_query": "agent-safety" + }, + { + "id": "2606.00644", + "title": "ForeSci: Evaluating LLM Agents for Forward-Looking AI Research Judgment", + "url": "https://arxiv.org/abs/2606.00644", + "published": "2026-05-30", + "updated": "2026-06-04", + "authors": [ + "Qiuyu Tian", + "Haojie Yin", + "Yingce Xia", + "Youyong Kong", + "Zequn Liu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2606.00644", + "source": "arxiv", + "source_id": "arxiv:2606.00644", + "pdf_url": "https://arxiv.org/pdf/2606.00644", + "primary_query": "rag-agent" + }, + { + "id": "2605.31308", + "title": "TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories", + "url": "https://arxiv.org/abs/2605.31308", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Junjie Nian", + "Kang Chen", + "Ge Zhang", + "Yixin Cao", + "Yugang Jiang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "embodied-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation" + ], + "arxiv_id": "2605.31308", + "source": "arxiv", + "source_id": "arxiv:2605.31308", + "pdf_url": "https://arxiv.org/pdf/2605.31308", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.31075", + "title": "Task-Focused Memorization for Multimodal Agents", + "url": "https://arxiv.org/abs/2605.31075", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Tao Zou", + "Yichen He", + "Tian Qiu", + "Yuan Lin", + "Hang Li" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.31075", + "source": "arxiv", + "source_id": "arxiv:2605.31075", + "pdf_url": "https://arxiv.org/pdf/2605.31075", + "primary_query": "agent-memory" + }, + { + "id": "2605.31268", + "title": "Mellum2 Technical Report", + "url": "https://arxiv.org/abs/2605.31268", + "published": "2026-05-29", + "updated": "2026-05-29", + "authors": [ + "Marko Kojic", + "Ivan Bondyrev", + "Aral de Moor", + "Joseph Shtok", + "Petr Borovlev", + "Kseniia Lysaniuk", + "Madeeswaran Kannan", + "Ivan Dolgov", + "Nikita Pavlichenko" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2605.31268", + "source": "arxiv", + "source_id": "arxiv:2605.31268", + "pdf_url": "https://arxiv.org/pdf/2605.31268", + "primary_query": "function-calling" + }, + { + "id": "2605.29653", + "title": "PTCG-Bench: Can LLM Agents Master Pokémon Trading Card Game?", + "url": "https://arxiv.org/abs/2605.29653", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Dongdong Hua", + "Yifei Sun", + "Renhong Huang", + "Feng Gao", + "Chunping Wang", + "Yang Yang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-evaluation", + "autonomous-agent-llm" + ], + "arxiv_id": "2605.29653", + "source": "arxiv", + "source_id": "arxiv:2605.29653", + "pdf_url": "https://arxiv.org/pdf/2605.29653", + "primary_query": "agent-evaluation" + }, + { + "id": "2605.29630", + "title": "Entity-Collision: A Stratified Protocol for Attributing Retrieval Lift in Agent Memory", + "url": "https://arxiv.org/abs/2605.29630", + "published": "2026-05-28", + "updated": "2026-05-28", + "authors": [ + "Youwang Deng" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-memory" + ], + "arxiv_id": "2605.29630", + "source": "arxiv", + "source_id": "arxiv:2605.29630", + "pdf_url": "https://arxiv.org/pdf/2605.29630", + "primary_query": "agent-memory" + }, + { + "id": "2606.07591", + "title": "ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research", + "url": "https://arxiv.org/abs/2606.07591", + "published": "2026-05-28", + "updated": "2026-07-03", + "authors": [ + "Wanghan Xu", + "Shuo Li", + "Tianlin Ye", + "Qinglong Cao", + "Yixin Chen", + "Hengjian Gao", + "Yiheng Wang", + "Qi Li", + "Kun Li", + "Sheng Xu", + "Shengdu Chai", + "Fangchen Yu", + "Xiangyu Zhao", + "Zhangrui Zhao", + "Weijie Ma", + "Zijie Guo", + "Koutian Wu", + "Haoyu Zhou", + "Haoxiang Yin", + "Lixue Cheng", + "Chaofan Hu", + "Haoxuan Li", + "Lu Mi", + "Xuxuan Xie", + "Yifan Zhou", + "Ruizhe Chen", + "Zhiwang Zhou", + "Xingjian Guo", + "Yuhao Zhou", + "Xuming He", + "Shengyuan Xu", + "Xinyu Gu", + "Jiamin Wu", + "Mianxin Liu", + "Chunfeng Song", + "Fenghua Ling", + "Dongzhan Zhou", + "Shixiang Tang", + "Yuqiang Li", + "Mao Su", + "Peng Ye", + "Siqi Sun", + "Bin Wang", + "Xue Yang", + "Zhenfei Yin", + "Tianfan Fu", + "Guangtao Zhai", + "Wanli Ouyang", + "Bo Zhang", + "Lei Bai", + "Wenlong Zhang" + ], + "categories": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2606.07591", + "source": "arxiv", + "source_id": "arxiv:2606.07591", + "pdf_url": "https://arxiv.org/pdf/2606.07591", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.28617", + "title": "LACUNA: Safe Agents as Recursive Program Holes", + "url": "https://arxiv.org/abs/2605.28617", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Yaoyu Zhao", + "Yichen Xu", + "Oliver Bračevac", + "Cao Nguyen Pham", + "Frank Zhengqing Wu", + "Martin Odersky" + ], + "categories": [ + "cs.AI", + "cs.PL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.28617", + "source": "arxiv", + "source_id": "arxiv:2605.28617", + "pdf_url": "https://arxiv.org/pdf/2605.28617", + "primary_query": "planning-agent" + }, + { + "id": "2605.28424", + "title": "Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning", + "url": "https://arxiv.org/abs/2605.28424", + "published": "2026-05-27", + "updated": "2026-05-27", + "authors": [ + "Jiapeng Zhu", + "Jianxiang Yu", + "Yibo Zhao", + "Chengcheng Han", + "Qi Gu", + "Xunliang Cai", + "Xiang Li", + "Weining Qian" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-safety", + "memory" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.28424", + "source": "arxiv", + "source_id": "arxiv:2605.28424", + "pdf_url": "https://arxiv.org/pdf/2605.28424", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.26497", + "title": "Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents", + "url": "https://arxiv.org/abs/2605.26497", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Peiran Wang", + "Ying Li", + "Yuan Tian" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2605.26497", + "source": "arxiv", + "source_id": "arxiv:2605.26497", + "pdf_url": "https://arxiv.org/pdf/2605.26497", + "primary_query": "agent-safety" + }, + { + "id": "2605.27123", + "title": "Rethinking Agentic RAG: Toward LLM-Driven Logical Retrieval Beyond Embeddings", + "url": "https://arxiv.org/abs/2605.27123", + "published": "2026-05-26", + "updated": "2026-05-26", + "authors": [ + "Yuqi Zeng", + "Qixiang Deng", + "Yulei Wan", + "Ruiquan Jiang", + "Xiaoqing Zheng", + "Xuanjing Huang" + ], + "categories": [ + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.27123", + "source": "arxiv", + "source_id": "arxiv:2605.27123", + "pdf_url": "https://arxiv.org/pdf/2605.27123", + "primary_query": "rag-agent" + }, + { + "id": "2605.26165", + "title": "Tool-Schema Compression Enables Agentic RAG Under Constrained Context Budgets", + "url": "https://arxiv.org/abs/2605.26165", + "published": "2026-05-24", + "updated": "2026-05-24", + "authors": [ + "Furkan Sakizli" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "rag-agent" + ], + "arxiv_id": "2605.26165", + "source": "arxiv", + "source_id": "arxiv:2605.26165", + "pdf_url": "https://arxiv.org/pdf/2605.26165", + "primary_query": "rag-agent" + }, + { + "id": "2605.23899", + "title": "From Raw Experience to Skill Consumption: A Systematic Study of Model-Generated Agent Skills", + "url": "https://arxiv.org/abs/2605.23899", + "published": "2026-05-22", + "updated": "2026-05-22", + "authors": [ + "Zisu Huang", + "Jingwen Xu", + "Yifan Yang", + "Ziyang Gong", + "Qihao Yang", + "Muzhao Tian", + "Xiaohua Wang", + "Changze Lv", + "Xuemei Gao", + "Qi Dai", + "Bei Liu", + "Kai Qiu", + "Xue Yang", + "Dongdong Chen", + "Xiaoqing Zheng", + "Chong Luo" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.23899", + "source": "arxiv", + "source_id": "arxiv:2605.23899", + "pdf_url": "https://arxiv.org/pdf/2605.23899", + "primary_query": "language-agent" + }, + { + "id": "2605.17075", + "title": "A Red Teaming Framework for Evaluating Robustness of AI-enabled Security Orchestration, Automation, and Response Systems", + "url": "https://arxiv.org/abs/2605.17075", + "published": "2026-05-16", + "updated": "2026-05-16", + "authors": [ + "Ayan Javeed Shaikh", + "Nathaniel D. Bastian", + "Ankit Shah" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.17075", + "source": "arxiv", + "source_id": "arxiv:2605.17075", + "pdf_url": "https://arxiv.org/pdf/2605.17075", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.14322", + "title": "Are Agents Ready to Teach? A Multi-Stage Benchmark for Real-World Teaching Workflows", + "url": "https://arxiv.org/abs/2605.14322", + "published": "2026-05-14", + "updated": "2026-05-20", + "authors": [ + "Zixin Chen", + "Peng Liu", + "Rui Sheng", + "Haobo Li", + "Jianhong Tu", + "Xiaodong Deng", + "Kashun Shum", + "Dayiheng Liu", + "Huamin Qu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.14322", + "source": "arxiv", + "source_id": "arxiv:2605.14322", + "pdf_url": "https://arxiv.org/pdf/2605.14322", + "primary_query": "language-agent" + }, + { + "id": "2605.11928", + "title": "When Simulation Lies: A Sim-to-Real Benchmark and Domain-Randomized RL Recipe for Tool-Use Agents", + "url": "https://arxiv.org/abs/2605.11928", + "published": "2026-05-12", + "updated": "2026-05-12", + "authors": [ + "Xiaolin Zhou", + "Aojie Yuan", + "Zheng Luo", + "Zipeng Ling", + "Xixiao Pan", + "Yicheng Gao", + "Haiyue Zhang", + "Jiate Li", + "Shuli Jiang", + "Prince Zizhuang Wang", + "Zixuan Zhu", + "Jinbo Liu", + "Ryan A. Rossi", + "Hua Wei", + "Xiyang Hu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling", + "language-agent" + ], + "arxiv_id": "2605.11928", + "source": "arxiv", + "source_id": "arxiv:2605.11928", + "pdf_url": "https://arxiv.org/pdf/2605.11928", + "primary_query": "function-calling" + }, + { + "id": "2605.10870", + "title": "Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory", + "url": "https://arxiv.org/abs/2605.10870", + "published": "2026-05-11", + "updated": "2026-05-11", + "authors": [ + "Mingxi Zou", + "Zhihan Guo", + "Langzhang Liang", + "Zhuo Wang", + "Qifan Wang", + "Qingsong Wen", + "Irwin King", + "Lizhen Qu", + "Zenglin Xu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "memory", + "planning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2605.10870", + "source": "arxiv", + "source_id": "arxiv:2605.10870", + "pdf_url": "https://arxiv.org/pdf/2605.10870", + "primary_query": "language-agent" + }, + { + "id": "2605.08876", + "title": "OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents", + "url": "https://arxiv.org/abs/2605.08876", + "published": "2026-05-09", + "updated": "2026-06-07", + "authors": [ + "Xinyu Li", + "Ronghui Mu", + "Lin Li", + "Tianjin Huang", + "Gaojie Jin" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "computer-use", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2605.08876", + "source": "arxiv", + "source_id": "arxiv:2605.08876", + "pdf_url": "https://arxiv.org/pdf/2605.08876", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2605.03328", + "title": "LLM-ADAM: A Generalizable LLM Agent Framework for Pre-Print Anomaly Detection in Additive Manufacturing", + "url": "https://arxiv.org/abs/2605.03328", + "published": "2026-05-05", + "updated": "2026-05-05", + "authors": [ + "Ahmadreza Eslaminia", + "Chuhan Cai", + "Cameron Smith", + "Ruo-Syuan Mei", + "Shichen Li", + "Rajiv Malhotra", + "Klara Nahrstedt", + "Chenhui Shao" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "planning-agent" + ], + "arxiv_id": "2605.03328", + "source": "arxiv", + "source_id": "arxiv:2605.03328", + "pdf_url": "https://arxiv.org/pdf/2605.03328", + "primary_query": "planning-agent" + }, + { + "id": "2604.27092", + "title": "End-to-end autonomous scientific discovery on a real optical platform", + "url": "https://arxiv.org/abs/2604.27092", + "published": "2026-04-29", + "updated": "2026-04-29", + "authors": [ + "Shuxing Yang", + "Fujia Chen", + "Rui Zhao", + "Junyao Wu", + "Yize Wang", + "Haiyao Luo", + "Ning Han", + "Qiaolu Chen", + "Yuze Hu", + "Wenhao Li", + "Mingzhu Li", + "Hongsheng Chen", + "Yihao Yang" + ], + "categories": [ + "cs.AI", + "physics.optics" + ], + "topics": [ + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "autonomous-agent-llm" + ], + "arxiv_id": "2604.27092", + "source": "arxiv", + "source_id": "arxiv:2604.27092", + "pdf_url": "https://arxiv.org/pdf/2604.27092", + "primary_query": "autonomous-agent-llm" + }, + { + "id": "2604.20994", + "title": "Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models", + "url": "https://arxiv.org/abs/2604.20994", + "published": "2026-04-22", + "updated": "2026-04-22", + "authors": [ + "Yannis Belkhiter", + "Giulio Zizzo", + "Sergio Maffeis", + "Seshu Tirupathi", + "John D. Kelleher" + ], + "categories": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2604.20994", + "source": "arxiv", + "source_id": "arxiv:2604.20994", + "pdf_url": "https://arxiv.org/pdf/2604.20994", + "primary_query": "function-calling" + }, + { + "id": "2603.28900", + "title": "Robust Multi-Agent Reinforcement Learning for Small UAS Separation Assurance under GPS Degradation and Spoofing", + "url": "https://arxiv.org/abs/2603.28900", + "published": "2026-03-30", + "updated": "2026-03-30", + "authors": [ + "Alex Zongo", + "Filippos Fotiadis", + "Ufuk Topcu", + "Peng Wei" + ], + "categories": [ + "cs.RO", + "cs.AI", + "cs.LG", + "eess.SY" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.28900", + "source": "arxiv", + "source_id": "arxiv:2603.28900", + "pdf_url": "https://arxiv.org/pdf/2603.28900", + "primary_query": "agent-safety" + }, + { + "id": "2603.25353", + "title": "SafeGuard ASF: SR Agentic Humanoid Robot System for Autonomous Industrial Safety", + "url": "https://arxiv.org/abs/2603.25353", + "published": "2026-03-26", + "updated": "2026-03-26", + "authors": [ + "Thanh Nguyen Canh", + "Thang Tran Viet", + "Thanh Tuan Tran", + "Ben Wei Lim" + ], + "categories": [ + "cs.RO" + ], + "topics": [ + "agent-safety", + "embodied-agent", + "reasoning", + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.25353", + "source": "arxiv", + "source_id": "arxiv:2603.25353", + "pdf_url": "https://arxiv.org/pdf/2603.25353", + "primary_query": "agent-safety" + }, + { + "id": "2603.19684", + "title": "TSegAgent: Zero-Shot Tooth Segmentation via Geometry-Aware Vision-Language Agents", + "url": "https://arxiv.org/abs/2603.19684", + "published": "2026-03-20", + "updated": "2026-06-23", + "authors": [ + "Shaojie Zhuang", + "Lu Yin", + "Guangshun Wei", + "Yunpeng Li", + "Xilu Wang", + "Yuanfeng Zhou" + ], + "categories": [ + "cs.CV" + ], + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.19684", + "source": "arxiv", + "source_id": "arxiv:2603.19684", + "pdf_url": "https://arxiv.org/pdf/2603.19684", + "primary_query": "language-agent" + }, + { + "id": "2603.17392", + "title": "Agentic Cognitive Profiling: Realigning Automated Alzheimer's Disease Detection with Clinical Construct Validity", + "url": "https://arxiv.org/abs/2603.17392", + "published": "2026-03-18", + "updated": "2026-03-18", + "authors": [ + "Jiawen Kang", + "Kun Li", + "Dongrui Han", + "Jinchao Li", + "Junan Li", + "Lingwei Meng", + "Xixin Wu", + "Helen Meng" + ], + "categories": [ + "cs.MA", + "cs.IR", + "q-bio.NC" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2603.17392", + "source": "arxiv", + "source_id": "arxiv:2603.17392", + "pdf_url": "https://arxiv.org/pdf/2603.17392", + "primary_query": "function-calling" + }, + { + "id": "2603.15666", + "title": "Compiled Memory: Not More Information, but More Precise Instructions for Language Agents", + "url": "https://arxiv.org/abs/2603.15666", + "published": "2026-03-12", + "updated": "2026-03-12", + "authors": [ + "James Rhodes", + "George Kang" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "memory", + "rag" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.15666", + "source": "arxiv", + "source_id": "arxiv:2603.15666", + "pdf_url": "https://arxiv.org/pdf/2603.15666", + "primary_query": "language-agent" + }, + { + "id": "2603.00801", + "title": "The Synthetic Web: Adversarially-Curated Mini-Internets for Diagnosing Epistemic Weaknesses of Language Agents", + "url": "https://arxiv.org/abs/2603.00801", + "published": "2026-02-28", + "updated": "2026-02-28", + "authors": [ + "Shrey Shah", + "Levent Ozgur" + ], + "categories": [ + "cs.AI", + "cs.IR" + ], + "topics": [ + "agent-evaluation", + "embodied-agent", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2603.00801", + "source": "arxiv", + "source_id": "arxiv:2603.00801", + "pdf_url": "https://arxiv.org/pdf/2603.00801", + "primary_query": "language-agent" + }, + { + "id": "2602.21127", + "title": "\"Are You Sure?\": An Empirical Study of Human Perception Vulnerability in LLM-Driven Agentic Systems", + "url": "https://arxiv.org/abs/2602.21127", + "published": "2026-02-24", + "updated": "2026-02-24", + "authors": [ + "Xinfeng Li", + "Shenyu Dai", + "Kelong Zheng", + "Yue Xiao", + "Gelei Deng", + "Wei Dong", + "Xiaofeng Wang" + ], + "categories": [ + "cs.HC", + "cs.AI", + "cs.CR", + "cs.SI" + ], + "topics": [ + "agent-safety", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.21127", + "source": "arxiv", + "source_id": "arxiv:2602.21127", + "pdf_url": "https://arxiv.org/pdf/2602.21127", + "primary_query": "agent-safety" + }, + { + "id": "2603.00131", + "title": "Thought Virus: Viral Misalignment via Subliminal Prompting in Multi-Agent Systems", + "url": "https://arxiv.org/abs/2603.00131", + "published": "2026-02-23", + "updated": "2026-02-23", + "authors": [ + "Moritz Weckbecker", + "Jonas Müller", + "Ben Hagag", + "Michael Mulet" + ], + "categories": [ + "cs.MA", + "cs.AI" + ], + "topics": [ + "agent-safety", + "multi-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2603.00131", + "source": "arxiv", + "source_id": "arxiv:2603.00131", + "pdf_url": "https://arxiv.org/pdf/2603.00131", + "primary_query": "agent-safety" + }, + { + "id": "2602.19008", + "title": "Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks", + "url": "https://arxiv.org/abs/2602.19008", + "published": "2026-02-22", + "updated": "2026-02-22", + "authors": [ + "Wilson Y. Lee" + ], + "categories": [ + "cs.CL", + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "language-agent" + ], + "arxiv_id": "2602.19008", + "source": "arxiv", + "source_id": "arxiv:2602.19008", + "pdf_url": "https://arxiv.org/pdf/2602.19008", + "primary_query": "language-agent" + }, + { + "id": "2602.14234", + "title": "REDSearcher: A Scalable and Cost-Efficient Framework for Long-Horizon Search Agents", + "url": "https://arxiv.org/abs/2602.14234", + "published": "2026-02-15", + "updated": "2026-02-15", + "authors": [ + "Zheng Chu", + "Xiao Wang", + "Jack Hong", + "Huiming Fan", + "Yuqi Huang", + "Yue Yang", + "Guohai Xu", + "Chenxiao Zhao", + "Cheng Xiang", + "Shengchao Hu", + "Dongdong Kuang", + "Ming Liu", + "Bing Qin", + "Xing Yu" + ], + "categories": [ + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2602.14234", + "source": "arxiv", + "source_id": "arxiv:2602.14234", + "pdf_url": "https://arxiv.org/pdf/2602.14234", + "primary_query": "function-calling" + }, + { + "id": "2602.14281", + "title": "MCPShield: A Security Cognition Layer for Adaptive Trust Calibration in Model Context Protocol Agents", + "url": "https://arxiv.org/abs/2602.14281", + "published": "2026-02-15", + "updated": "2026-02-24", + "authors": [ + "Zhenhong Zhou", + "Yuanhe Zhang", + "Hongwei Cai", + "Moayad Aloqaily", + "Ouns Bouachir", + "Linsey Pang", + "Prakhar Mehrotra", + "Kun Wang", + "Qingsong Wen" + ], + "categories": [ + "cs.CR", + "cs.CL" + ], + "topics": [ + "agent-safety", + "computer-use", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.14281", + "source": "arxiv", + "source_id": "arxiv:2602.14281", + "pdf_url": "https://arxiv.org/pdf/2602.14281", + "primary_query": "agent-safety" + }, + { + "id": "2602.13665", + "title": "HyFunc: Accelerating LLM-based Function Calls for Agentic AI through Hybrid-Model Cascade and Dynamic Templating", + "url": "https://arxiv.org/abs/2602.13665", + "published": "2026-02-14", + "updated": "2026-02-14", + "authors": [ + "Weibin Liao", + "Jian-guang Lou", + "Haoyi Xiong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2602.13665", + "source": "arxiv", + "source_id": "arxiv:2602.13665", + "pdf_url": "https://arxiv.org/pdf/2602.13665", + "primary_query": "function-calling" + }, + { + "id": "2602.08082", + "title": "Spectral Guardrails for Agents in the Wild: Detecting Tool Use Hallucinations via Attention Topology", + "url": "https://arxiv.org/abs/2602.08082", + "published": "2026-02-08", + "updated": "2026-02-08", + "authors": [ + "Valentin Noël" + ], + "categories": [ + "cs.LG", + "cs.AI", + "eess.SP" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.08082", + "source": "arxiv", + "source_id": "arxiv:2602.08082", + "pdf_url": "https://arxiv.org/pdf/2602.08082", + "primary_query": "agent-safety" + }, + { + "id": "2602.10133", + "title": "AgentTrace: A Structured Logging Framework for Agent System Observability", + "url": "https://arxiv.org/abs/2602.10133", + "published": "2026-02-07", + "updated": "2026-02-07", + "authors": [ + "Adam AlSayyad", + "Kelvin Yuxiang Huang", + "Richik Pal" + ], + "categories": [ + "cs.SE", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.10133", + "source": "arxiv", + "source_id": "arxiv:2602.10133", + "pdf_url": "https://arxiv.org/pdf/2602.10133", + "primary_query": "agent-safety" + }, + { + "id": "2602.05386", + "title": "Spider-Sense: Intrinsic Risk Sensing for Efficient Agent Defense with Hierarchical Adaptive Screening", + "url": "https://arxiv.org/abs/2602.05386", + "published": "2026-02-05", + "updated": "2026-02-06", + "authors": [ + "Zhenxiong Yu", + "Zhi Yang", + "Zhiheng Jin", + "Shuhe Wang", + "Heng Zhang", + "Yanlin Fei", + "Lingfeng Zeng", + "Fangqi Lou", + "Shuo Zhang", + "Tu Hu", + "Jingping Liu", + "Rongze Chen", + "Xingyu Zhu", + "Kunyi Wang", + "Chaofa Yuan", + "Xin Guo", + "Zhaowei Liu", + "Feipeng Zhang", + "Jie Huang", + "Huacan Wang", + "Ronghao Chen", + "Liwen Zhang" + ], + "categories": [ + "cs.CR", + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.05386", + "source": "arxiv", + "source_id": "arxiv:2602.05386", + "pdf_url": "https://arxiv.org/pdf/2602.05386", + "primary_query": "agent-safety" + }, + { + "id": "2602.03117", + "title": "AgentDyn: Are Your Agent Security Defenses Deployable in Real-World Dynamic Environments?", + "url": "https://arxiv.org/abs/2602.03117", + "published": "2026-02-03", + "updated": "2026-05-07", + "authors": [ + "Hao Li", + "Ruoyao Wen", + "Shanghao Shi", + "Ning Zhang", + "Yevgeniy Vorobeychik", + "Chaowei Xiao" + ], + "categories": [ + "cs.CR" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "agent-safety" + ], + "arxiv_id": "2602.03117", + "source": "arxiv", + "source_id": "arxiv:2602.03117", + "pdf_url": "https://arxiv.org/pdf/2602.03117", + "primary_query": "agent-safety" + }, + { + "id": "2601.12988", + "title": "PaperGuide: Making Small Language-Model Paper-Reading Agents More Efficient", + "url": "https://arxiv.org/abs/2601.12988", + "published": "2026-01-19", + "updated": "2026-01-19", + "authors": [ + "Zijian Wang", + "Tiancheng Huang", + "Hanqi Li", + "Da Ma", + "Lu Chen", + "Kai Yu" + ], + "categories": [ + "cs.LG" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.12988", + "source": "arxiv", + "source_id": "arxiv:2601.12988", + "pdf_url": "https://arxiv.org/pdf/2601.12988", + "primary_query": "function-calling" + }, + { + "id": "2601.06606", + "title": "CEDAR: Context Engineering for Agentic Data Science", + "url": "https://arxiv.org/abs/2601.06606", + "published": "2026-01-10", + "updated": "2026-04-22", + "authors": [ + "Rishiraj Saha Roy", + "Chris Hinze", + "Luzian Hahn", + "Fabian Kuech" + ], + "categories": [ + "cs.LG", + "cs.AI" + ], + "topics": [ + "planning", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2601.06606", + "source": "arxiv", + "source_id": "arxiv:2601.06606", + "pdf_url": "https://arxiv.org/pdf/2601.06606", + "primary_query": "function-calling" + }, + { + "id": "2512.23747", + "title": "State-of-the-art Small Language Coder Model: Mify-Coder", + "url": "https://arxiv.org/abs/2512.23747", + "published": "2025-12-26", + "updated": "2025-12-26", + "authors": [ + "Abhinav Parmar", + "Abhisek Panigrahi", + "Abhishek Kumar Dwivedi", + "Abhishek Bhattacharya", + "Adarsh Ramachandra", + "Aditya Choudhary", + "Aditya Garg", + "Aditya Raj", + "Alankrit Bhatt", + "Alpesh Yadav", + "Anant Vishnu", + "Ananthu Pillai", + "Ankush Kumar", + "Aryan Patnaik", + "Aswatha Narayanan S", + "Avanish Raj Singh", + "Bhavya Shree Gadda", + "Brijesh Pankajbhai Kachhadiya", + "Buggala Jahnavi", + "Chidurala Nithin Krishna", + "Chintan Shah", + "Chunduru Akshaya", + "Debarshi Banerjee", + "Debrup Dey", + "Deepa R.", + "Deepika B G", + "Faiz ur Rahman", + "Gagan Gayari", + "Gudhi Jagadeesh Kumar Naidu", + "Gursimar Singh", + "Harshal Tyagi", + "Harshini K", + "James Mani Vathalloor", + "Jayarama Nettar", + "Jayashree Gajjam", + "Joe Walter Sugil George", + "Kamalakara Sri Krishna Tadepalli", + "Kamalkumar Rathinasamy", + "Karan Chaurasia", + "Karthikeyan S", + "Kashish Arora", + "Kaushal Desai", + "Khushboo Buwade", + "Kiran Manjrekar", + "Malikireddy Venkata Sai Likhitha", + "Manjunath A", + "Mitali Mahavir Bedmutha", + "Mohammed Rafee Tarafdar", + "Nikhil Tiwari", + "Nikitha K Gigi", + "Pavan Ravikumar", + "Pendyala Swarnanjali", + "Piyush Anand", + "Prakash Chandrasekar", + "Prasanna Bhalchandra Gawade", + "Prasanth Sivan", + "Preeti Khurana", + "Priyanshi Babbar", + "Rajab Ali Mondal", + "Rajesh Kumar Vissapragada", + "Rajeshwari Ganesan", + "Rajeswari Koppisetti", + "Ramjee R.", + "Ramkumar Thiruppathisamy", + "Rani G. S.", + "S Reka", + "Samarth Gupta", + "Sandeep Reddy Kothakota", + "Sarathy K", + "Sathyanarayana Sampath Kumar", + "Saurabh Kumar", + "Shashank Khasare", + "Shenbaga Devi Venkatesh Kumar", + "Shiva Rama Krishna Parvatham", + "Shoeb Shaikh", + "Shrishanmathi A", + "Shubham Pathak", + "Sree Samhita Koppaka", + "Sreenivasa Raghavan K S", + "Sreeram Venkatasubramanian", + "Suprabha Desai Bojja", + "Swetha R", + "Syed Ahmed", + "Chinmai Harshitha Thota", + "Tushar Yadav", + "Veeravelly Kusumitha", + "V V S S Prasanth Patnaik", + "Vidya Sri Sesetti", + "Vijayakeerthi K", + "Vikram Raj Bakshi", + "Vinay K K", + "Vinoth Kumar Loganathan", + "Vipin Tiwari", + "Vivek Kumar Shrivastav", + "V Venkata Sri Datta Charan", + "Wasim Akhtar Khan" + ], + "categories": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2512.23747", + "source": "arxiv", + "source_id": "arxiv:2512.23747", + "pdf_url": "https://arxiv.org/pdf/2512.23747", + "primary_query": "function-calling" + }, + { + "id": "2510.24645", + "title": "FunReason-MT Technical Report: Advanced Data Synthesis Solution for Real-world Multi-Turn Tool-use", + "url": "https://arxiv.org/abs/2510.24645", + "published": "2025-10-28", + "updated": "2025-11-16", + "authors": [ + "Zengzhuang Xu", + "Bingguang Hao", + "Zechuan Wang", + "Yuntao Wen", + "Xinyi Xu", + "Yang Liu", + "Long Chen", + "Dong Wang", + "Maolin Wang", + "Tong Zhao", + "Yicheng Chen", + "Cunyin Peng", + "Jinjie Gu", + "Leilei Gan", + "Xiangyu Zhao", + "Chenyi Zhuang", + "Shi Gu" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.24645", + "source": "arxiv", + "source_id": "arxiv:2510.24645", + "pdf_url": "https://arxiv.org/pdf/2510.24645", + "primary_query": "function-calling" + }, + { + "id": "2510.04206", + "title": "AgentRL: Scaling Agentic Reinforcement Learning with a Multi-Turn, Multi-Task Framework", + "url": "https://arxiv.org/abs/2510.04206", + "published": "2025-10-05", + "updated": "2025-10-05", + "authors": [ + "Hanchen Zhang", + "Xiao Liu", + "Bowen Lv", + "Xueqiao Sun", + "Bohao Jing", + "Iat Long Iong", + "Zhenyu Hou", + "Zehan Qi", + "Hanyu Lai", + "Yifan Xu", + "Rui Lu", + "Hongning Wang", + "Jie Tang", + "Yuxiao Dong" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2510.04206", + "source": "arxiv", + "source_id": "arxiv:2510.04206", + "pdf_url": "https://arxiv.org/pdf/2510.04206", + "primary_query": "function-calling" + }, + { + "id": "2509.13311", + "title": "Towards General Agentic Intelligence via Environment Scaling", + "url": "https://arxiv.org/abs/2509.13311", + "published": "2025-09-16", + "updated": "2025-09-16", + "authors": [ + "Runnan Fang", + "Shihao Cai", + "Baixuan Li", + "Jialong Wu", + "Guangyu Li", + "Wenbiao Yin", + "Xinyu Wang", + "Xiaobin Wang", + "Liangcai Su", + "Zhen Zhang", + "Shibin Wu", + "Zhengwei Tao", + "Yong Jiang", + "Pengjun Xie", + "Fei Huang", + "Jingren Zhou" + ], + "categories": [ + "cs.CL" + ], + "topics": [ + "agent-evaluation", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.13311", + "source": "arxiv", + "source_id": "arxiv:2509.13311", + "pdf_url": "https://arxiv.org/pdf/2509.13311", + "primary_query": "function-calling" + }, + { + "id": "2509.02494", + "title": "GridMind: LLMs-Powered Agents for Power System Analysis and Operations", + "url": "https://arxiv.org/abs/2509.02494", + "published": "2025-09-02", + "updated": "2025-09-02", + "authors": [ + "Hongwei Jin", + "Kibaek Kim", + "Jonghwan Kwon" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "multi-agent", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2509.02494", + "source": "arxiv", + "source_id": "arxiv:2509.02494", + "pdf_url": "https://arxiv.org/pdf/2509.02494", + "primary_query": "function-calling" + }, + { + "id": "2508.17094", + "title": "PowerChain: A Verifiable Agentic AI System for Automating Distribution Grid Analyses", + "url": "https://arxiv.org/abs/2508.17094", + "published": "2025-08-23", + "updated": "2025-10-21", + "authors": [ + "Emmanuel O. Badmus", + "Peng Sang", + "Dimitrios Stamoulis", + "Amritanshu Pandey" + ], + "categories": [ + "cs.AI", + "eess.SY" + ], + "topics": [ + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.17094", + "source": "arxiv", + "source_id": "arxiv:2508.17094", + "pdf_url": "https://arxiv.org/pdf/2508.17094", + "primary_query": "function-calling" + }, + { + "id": "2508.12685", + "title": "ToolACE-MT: Non-Autoregressive Generation for Agentic Multi-Turn Interaction", + "url": "https://arxiv.org/abs/2508.12685", + "published": "2025-08-18", + "updated": "2026-02-13", + "authors": [ + "Xingshan Zeng", + "Weiwen Liu", + "Lingzhi Wang", + "Liangyou Li", + "Fei Mi", + "Yasheng Wang", + "Lifeng Shang", + "Xin Jiang", + "Qun Liu" + ], + "categories": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "topics": [ + "tool-use", + "world-model" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2508.12685", + "source": "arxiv", + "source_id": "arxiv:2508.12685", + "pdf_url": "https://arxiv.org/pdf/2508.12685", + "primary_query": "function-calling" + }, + { + "id": "2507.20666", + "title": "MIMII-Agent: Leveraging LLMs with Function Calling for Relative Evaluation of Anomalous Sound Detection", + "url": "https://arxiv.org/abs/2507.20666", + "published": "2025-07-28", + "updated": "2025-07-28", + "authors": [ + "Harsh Purohit", + "Tomoya Nishida", + "Kota Dohi", + "Takashi Endo", + "Yohei Kawaguchi" + ], + "categories": [ + "eess.AS", + "cs.AI", + "cs.LG", + "cs.SD" + ], + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2507.20666", + "source": "arxiv", + "source_id": "arxiv:2507.20666", + "pdf_url": "https://arxiv.org/pdf/2507.20666", + "primary_query": "function-calling" + }, + { + "id": "2507.20395", + "title": "MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models", + "url": "https://arxiv.org/abs/2507.20395", + "published": "2025-07-27", + "updated": "2025-07-27", + "authors": [ + "Hafsteinn Einarsson" + ], + "categories": [ + "cs.AI" + ], + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "reasoning" + ], + "score": 13, + "relevance": "high", + "matched_queries": [ + "function-calling" + ], + "arxiv_id": "2507.20395", + "source": "arxiv", + "source_id": "arxiv:2507.20395", + "pdf_url": "https://arxiv.org/pdf/2507.20395", + "primary_query": "function-calling" + } +] diff --git a/data/index.json b/data/index.json index 3a53aa3..6252b01 100644 --- a/data/index.json +++ b/data/index.json @@ -301,6 +301,1245 @@ "relevance": "medium" } }, + { + "collection": "papers", + "path": "papers/items/2025-2507-20395-mazeeval-a-benchmark-for-testing-sequential-decision-making-in-language-models.md", + "title": "\"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models\"", + "authors": "Hafsteinn Einarsson", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2507.20395", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-07-27", + "updated_at": "2025-07-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2507-20666-mimii-agent-leveraging-llms-with-function-calling-for-relative-evaluation-of-ano.md", + "title": "\"MIMII-Agent: Leveraging LLMs with Function Calling for Relative Evaluation of Anomalous Sound Detection\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MIMII-Agent: Leveraging LLMs with Function Calling for Relative Evaluation of Anomalous Sound Detection\"", + "authors": "Harsh Purohit, Tomoya Nishida, Kota Dohi, Takashi Endo, Yohei Kawaguchi", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2507.20666", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-07-28", + "updated_at": "2025-07-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.AS", + "cs.AI", + "cs.LG", + "cs.SD" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2508-07575-mcptoolbench-a-large-scale-ai-agent-model-context-protocol-mcp-tool-use-benchmar.md", + "title": "\"MCPToolBench++: A Large Scale AI Agent Model Context Protocol MCP Tool Use Benchmark\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MCPToolBench++: A Large Scale AI Agent Model Context Protocol MCP Tool Use Benchmark\"", + "authors": "Shiqing Fan, Xichen Ding, Liang Zhang, Linjian Mo", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2508.07575", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-08-11", + "updated_at": "2025-08-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2508-11027-hell-or-high-water-evaluating-agentic-recovery-from-external-failures.md", + "title": "\"Hell or High Water: Evaluating Agentic Recovery from External Failures\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Hell or High Water: Evaluating Agentic Recovery from External Failures\"", + "authors": "Andrew Wang, Sophia Hager, Adi Asija, Daniel Khashabi, Nicholas Andrews", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2508.11027", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-08-14", + "updated_at": "2025-08-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2508-12685-toolace-mt-non-autoregressive-generation-for-agentic-multi-turn-interaction.md", + "title": "\"ToolACE-MT: Non-Autoregressive Generation for Agentic Multi-Turn Interaction\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ToolACE-MT: Non-Autoregressive Generation for Agentic Multi-Turn Interaction\"", + "authors": "Xingshan Zeng, Weiwen Liu, Lingzhi Wang, Liangyou Li, Fei Mi, Yasheng Wang, Lifeng Shang, Xin Jiang, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2508.12685", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-08-18", + "updated_at": "2026-02-13", + "status": "queued", + "relevance": "high", + "topics": [ + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2508-17094-powerchain-a-verifiable-agentic-ai-system-for-automating-distribution-grid-analy.md", + "title": "\"PowerChain: A Verifiable Agentic AI System for Automating Distribution Grid Analyses\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PowerChain: A Verifiable Agentic AI System for Automating Distribution Grid Analyses\"", + "authors": "Emmanuel O. Badmus, Peng Sang, Dimitrios Stamoulis, Amritanshu Pandey", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2508.17094", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-08-23", + "updated_at": "2025-10-21", + "status": "queued", + "relevance": "high", + "topics": [ + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2509-02444-appcopilot-toward-general-accurate-long-horizon-and-efficient-mobile-agent.md", + "title": "\"AppCopilot: Toward General, Accurate, Long-Horizon, and Efficient Mobile Agent\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AppCopilot: Toward General, Accurate, Long-Horizon, and Efficient Mobile Agent\"", + "authors": "Jingru Fan, Yufan Dang, Jingyao Wu, Huatao Li, Runde Yang, Xiyuan Yang, Yuheng Wang, Chen Qian", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2509.02444", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-09-02", + "updated_at": "2025-10-17", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "memory", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.CV", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2509-02494-gridmind-llms-powered-agents-for-power-system-analysis-and-operations.md", + "title": "\"GridMind: LLMs-Powered Agents for Power System Analysis and Operations\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GridMind: LLMs-Powered Agents for Power System Analysis and Operations\"", + "authors": "Hongwei Jin, Kibaek Kim, Jonghwan Kwon", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2509.02494", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-09-02", + "updated_at": "2025-09-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2509-08863-geojson-agents-a-multi-agent-llm-architecture-for-geospatial-analysis-function-c.md", + "title": "\"GeoJSON Agents:A Multi-Agent LLM Architecture for Geospatial Analysis-Function Calling vs Code Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GeoJSON Agents:A Multi-Agent LLM Architecture for Geospatial Analysis-Function Calling vs Code Generation\"", + "authors": "Qianqian Luo, Qingming Lin, Liuchang Xu, Sensen Wu, Ruichen Mao, Chao Wang, Hailin Feng, Bo Huang, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2509.08863", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-09-10", + "updated_at": "2025-12-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2509-10769-agentarch-a-comprehensive-benchmark-to-evaluate-agent-architectures-in-enterpris.md", + "title": "\"AgentArch: A Comprehensive Benchmark to Evaluate Agent Architectures in Enterprise\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentArch: A Comprehensive Benchmark to Evaluate Agent Architectures in Enterprise\"", + "authors": "Tara Bogavelli, Roshnee Sharma, Hari Subramani", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2509.10769", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-09-13", + "updated_at": "2026-01-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2509-13311-towards-general-agentic-intelligence-via-environment-scaling.md", + "title": "Towards General Agentic Intelligence via Environment Scaling", + "type": "paper", + "meta": { + "type": "paper", + "title": "Towards General Agentic Intelligence via Environment Scaling", + "authors": "Runnan Fang, Shihao Cai, Baixuan Li, Jialong Wu, Guangyu Li, Wenbiao Yin, Xinyu Wang, Xiaobin Wang, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2509.13311", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-09-16", + "updated_at": "2025-09-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2509-14477-ticket-bench-a-kickoff-for-multilingual-and-regionalized-agent-evaluation.md", + "title": "\"Ticket-Bench: A Kickoff for Multilingual and Regionalized Agent Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Ticket-Bench: A Kickoff for Multilingual and Regionalized Agent Evaluation\"", + "authors": "Thales Sales Almeida, João Guilherme Alves Santos, Thiago Laitz, Giovana Kerche Bonás", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2509.14477", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-09-17", + "updated_at": "2025-09-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2509-20998-core-full-path-evaluation-of-llm-agents-beyond-final-state.md", + "title": "\"CORE: Full-Path Evaluation of LLM Agents Beyond Final State\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CORE: Full-Path Evaluation of LLM Agents Beyond Final State\"", + "authors": "Panagiotis Michelakis, Yiannis Hadjiyiannis, Dimitrios Stamoulis", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2509.20998", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-09-25", + "updated_at": "2025-09-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2509-26553-towards-reliable-benchmarking-a-contamination-free-controllable-evaluation-frame.md", + "title": "\"Towards Reliable Benchmarking: A Contamination Free, Controllable Evaluation Framework for Multi-step LLM Function Calling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Towards Reliable Benchmarking: A Contamination Free, Controllable Evaluation Framework for Multi-step LLM Function Calling\"", + "authors": "Seiji Maekawa, Jackson Hassell, Pouya Pezeshkpour, Tom Mitchell, Estevam Hruschka", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2509.26553", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-09-30", + "updated_at": "2026-02-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.PL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2510-03847-small-language-models-for-agentic-systems-a-survey-of-architectures-capabilities.md", + "title": "\"Small Language Models for Agentic Systems: A Survey of Architectures, Capabilities, and Deployment Trade offs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Small Language Models for Agentic Systems: A Survey of Architectures, Capabilities, and Deployment Trade offs\"", + "authors": "Raghav Sharma, Manan Mehta", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2510.03847", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-10-04", + "updated_at": "2025-10-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2510-04206-agentrl-scaling-agentic-reinforcement-learning-with-a-multi-turn-multi-task-fram.md", + "title": "\"AgentRL: Scaling Agentic Reinforcement Learning with a Multi-Turn, Multi-Task Framework\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentRL: Scaling Agentic Reinforcement Learning with a Multi-Turn, Multi-Task Framework\"", + "authors": "Hanchen Zhang, Xiao Liu, Bowen Lv, Xueqiao Sun, Bohao Jing, Iat Long Iong, Zhenyu Hou, Zehan Qi, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2510.04206", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-10-05", + "updated_at": "2025-10-05", + "status": "queued", + "relevance": "high", + "topics": [ + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2510-14548-llm-agents-beyond-utility-an-open-ended-perspective.md", + "title": "\"LLM Agents Beyond Utility: An Open-Ended Perspective\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LLM Agents Beyond Utility: An Open-Ended Perspective\"", + "authors": "Asen Nachkov, Xi Wang, Luc Van Gool", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2510.14548", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-10-16", + "updated_at": "2025-10-16", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2510-18586-tokencake-a-kv-cache-centric-serving-framework-for-llm-based-multi-agent-applica.md", + "title": "\"TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications\"", + "authors": "Zhuohang Bian, Feiyang Wu, Zhuoran Li, Teng Ma, Youwei Zhuo", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2510.18586", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-10-21", + "updated_at": "2026-05-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2510-21524-eu-agent-bench-measuring-illegal-behavior-of-llm-agents-under-eu-law.md", + "title": "\"EU-Agent-Bench: Measuring Illegal Behavior of LLM Agents Under EU Law\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EU-Agent-Bench: Measuring Illegal Behavior of LLM Agents Under EU Law\"", + "authors": "Ilija Lichkovski, Alexander Müller, Mariam Ibrahim, Tiwai Mhundwa", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2510.21524", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-10-24", + "updated_at": "2025-10-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2510-22768-seeing-is-believing-evaluating-vision-language-model-susceptibility-in-agent-to-.md", + "title": "Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion", + "type": "paper", + "meta": { + "type": "paper", + "title": "Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion", + "authors": "Haoyi Qiu, Yilun Zhou, Pranav Narayanan Venkit, Kung-Hsiang Huang, Jiaxin Zhang, Nanyun Peng, Chien-Sheng Wu", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2510.22768", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-10-26", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2510-24645-funreason-mt-technical-report-advanced-data-synthesis-solution-for-real-world-mu.md", + "title": "\"FunReason-MT Technical Report: Advanced Data Synthesis Solution for Real-world Multi-Turn Tool-use\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"FunReason-MT Technical Report: Advanced Data Synthesis Solution for Real-world Multi-Turn Tool-use\"", + "authors": "Zengzhuang Xu, Bingguang Hao, Zechuan Wang, Yuntao Wen, Xinyi Xu, Yang Liu, Long Chen, Dong Wang, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2510.24645", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-10-28", + "updated_at": "2025-11-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2510-26167-toolrm-towards-agentic-tool-use-reward-modeling.md", + "title": "\"ToolRM: Towards Agentic Tool-Use Reward Modeling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ToolRM: Towards Agentic Tool-Use Reward Modeling\"", + "authors": "Renhao Li, Jianhong Tu, Yang Su, Yantao Liu, Fei Huang, Hamid Alinejad-Rokny, Derek F. Wong, Junyang Lin, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2510.26167", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-10-30", + "updated_at": "2026-01-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2511-04847-test-time-adaptation-for-llm-agents-via-environment-interaction.md", + "title": "Test-Time Adaptation for LLM Agents via Environment Interaction", + "type": "paper", + "meta": { + "type": "paper", + "title": "Test-Time Adaptation for LLM Agents via Environment Interaction", + "authors": "Arthur Chen, Zuxin Liu, Jianguo Zhang, Akshara Prabhakar, Zhiwei Liu, Shelby Heinecke, Silvio Savarese, Victor Zhong, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2511.04847", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-11-06", + "updated_at": "2026-02-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2511-11169-refine-and-align-confidence-calibration-through-multi-agent-interaction-in-vqa.md", + "title": "\"Refine and Align: Confidence Calibration through Multi-Agent Interaction in VQA\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Refine and Align: Confidence Calibration through Multi-Agent Interaction in VQA\"", + "authors": "Ayush Pandey, Jai Bardhan, Ishita Jain, Ramya S Hebbalaguppe, Rohan Raju Dhanakshirur, Lovekesh Vig", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2511.11169", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-11-14", + "updated_at": "2025-11-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2511-15203-taxonomy-evaluation-and-exploitation-of-ipi-centric-llm-agent-defense-frameworks.md", + "title": "Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks", + "type": "paper", + "meta": { + "type": "paper", + "title": "Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks", + "authors": "Zimo Ji, Xunguang Wang, Zongjie Li, Pingchuan Ma, Yudong Gao, Daoyuan Wu, Xincheng Yan, Tian Tian, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2511.15203", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-11-19", + "updated_at": "2025-11-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2511-22138-tinyllm-evaluation-and-optimization-of-small-language-models-for-agentic-tasks-o.md", + "title": "\"TinyLLM: Evaluation and Optimization of Small Language Models for Agentic Tasks on Edge Devices\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TinyLLM: Evaluation and Optimization of Small Language Models for Agentic Tasks on Edge Devices\"", + "authors": "Mohd Ariful Haque, Fahad Rahman, Kishor Datta Gupta, Khalil Shujaee, Roy George", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2511.22138", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-11-27", + "updated_at": "2025-11-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2512-02605-iact-a-self-organizing-recursive-model-for-general-ai-agents-a-technical-white-p.md", + "title": "\"IACT: A Self-Organizing Recursive Model for General AI Agents: A Technical White Paper on the Architecture Behind kragent.ai\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"IACT: A Self-Organizing Recursive Model for General AI Agents: A Technical White Paper on the Architecture Behind kragent.ai\"", + "authors": "Pengju Lu", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2512.02605", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-12-02", + "updated_at": "2025-12-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2512-11682-medai-evaluating-txagent-s-therapeutic-agentic-reasoning-in-the-neurips-cure-ben.md", + "title": "\"MedAI: Evaluating TxAgent's Therapeutic Agentic Reasoning in the NeurIPS CURE-Bench Competition\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MedAI: Evaluating TxAgent's Therapeutic Agentic Reasoning in the NeurIPS CURE-Bench Competition\"", + "authors": "Tim Cofala, Christian Kalfar, Jingge Xiao, Johanna Schrader, Michelle Tang, Wolfgang Nejdl", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2512.11682", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-12-12", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2512-23611-close-the-loop-synthesizing-infinite-tool-use-data-via-multi-agent-role-playing.md", + "title": "\"Close the Loop: Synthesizing Infinite Tool-Use Data via Multi-Agent Role-Playing\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Close the Loop: Synthesizing Infinite Tool-Use Data via Multi-Agent Role-Playing\"", + "authors": "Yuwen Li, Wei Zhang, Zelong Huang, Mason Yang, Jiajun Wu, Shawn Guo, Huahao Hu, Lingyi Sun, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2512.23611", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-12-29", + "updated_at": "2025-12-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2512-23647-nested-browser-use-learning-for-agentic-information-seeking.md", + "title": "Nested Browser-Use Learning for Agentic Information Seeking", + "type": "paper", + "meta": { + "type": "paper", + "title": "Nested Browser-Use Learning for Agentic Information Seeking", + "authors": "Baixuan Li, Jialong Wu, Wenbiao Yin, Kuan Li, Zhongwang Zhang, Huifeng Yin, Zhengwei Tao, Liwen Zhang, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2512.23647", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-12-29", + "updated_at": "2025-12-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.IR", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2025-2512-23747-state-of-the-art-small-language-coder-model-mify-coder.md", + "title": "\"State-of-the-art Small Language Coder Model: Mify-Coder\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"State-of-the-art Small Language Coder Model: Mify-Coder\"", + "authors": "Abhinav Parmar, Abhisek Panigrahi, Abhishek Kumar Dwivedi, Abhishek Bhattacharya, Adarsh Ramachandra, Aditya Choudhary, Aditya Garg, Aditya Raj, et al.", + "year": "2025", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2512.23747", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2025-12-26", + "updated_at": "2025-12-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, { "collection": "papers", "path": "papers/items/2025-vijayvargiya-openagentsafety.md", @@ -345,6 +1584,37873 @@ "related_projects": [] } }, + { + "collection": "papers", + "path": "papers/items/2026-2601-00268-beyond-perfect-apis-a-comprehensive-evaluation-of-llm-agents-under-real-world-ap.md", + "title": "\"Beyond Perfect APIs: A Comprehensive Evaluation of LLM Agents Under Real-World API Complexity\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Perfect APIs: A Comprehensive Evaluation of LLM Agents Under Real-World API Complexity\"", + "authors": "Doyoung Kim, Zhiwei Ren, Jie Hao, Zhongkai Sun, Lichao Wang, Xiyao Ma, Zack Ye, Xu Han, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2601.00268", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-01-01", + "updated_at": "2026-01-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2601-05467-stelp-secure-transpilation-and-execution-of-llm-generated-programs.md", + "title": "\"STELP: Secure Transpilation and Execution of LLM-Generated Programs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"STELP: Secure Transpilation and Execution of LLM-Generated Programs\"", + "authors": "Swapnil Shinde, Sahil Wadhwa, Andy Luo, Akshay Gupta, Mohammad Shahed Sorower", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2601.05467", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-01-09", + "updated_at": "2026-01-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2601-06007-don-t-break-the-cache-an-evaluation-of-prompt-caching-for-long-horizon-agentic-t.md", + "title": "\"Don't Break the Cache: An Evaluation of Prompt Caching for Long-Horizon Agentic Tasks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Don't Break the Cache: An Evaluation of Prompt Caching for Long-Horizon Agentic Tasks\"", + "authors": "Elias Lumer, Faheem Nizar, Akshaya Jangiti, Kevin Frank, Anmol Gulati, Mandar Phadate, Vamse Kumar Subbiah", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2601.06007", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-01-09", + "updated_at": "2026-01-31", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2601-06606-cedar-context-engineering-for-agentic-data-science.md", + "title": "\"CEDAR: Context Engineering for Agentic Data Science\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CEDAR: Context Engineering for Agentic Data Science\"", + "authors": "Rishiraj Saha Roy, Chris Hinze, Luzian Hahn, Fabian Kuech", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2601.06606", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-01-10", + "updated_at": "2026-04-22", + "status": "queued", + "relevance": "high", + "topics": [ + "planning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2601-12988-paperguide-making-small-language-model-paper-reading-agents-more-efficient.md", + "title": "\"PaperGuide: Making Small Language-Model Paper-Reading Agents More Efficient\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PaperGuide: Making Small Language-Model Paper-Reading Agents More Efficient\"", + "authors": "Zijian Wang, Tiancheng Huang, Hanqi Li, Da Ma, Lu Chen, Kai Yu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2601.12988", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-01-19", + "updated_at": "2026-01-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2601-14652-mas-orchestra-understanding-and-improving-multi-agent-reasoning-through-holistic.md", + "title": "\"MAS-Orchestra: Understanding and Improving Multi-Agent Reasoning Through Holistic Orchestration and Controlled Benchmarks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MAS-Orchestra: Understanding and Improving Multi-Agent Reasoning Through Holistic Orchestration and Controlled Benchmarks\"", + "authors": "Zixuan Ke, Yifei Ming, Austin Xu, Ryan Chin, Xuan-Phi Nguyen, Prathyusha Jwalapuram, Jiayu Wang, Semih Yavuz, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2601.14652", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-01-21", + "updated_at": "2026-05-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-03117-agentdyn-are-your-agent-security-defenses-deployable-in-real-world-dynamic-envir.md", + "title": "\"AgentDyn: Are Your Agent Security Defenses Deployable in Real-World Dynamic Environments?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentDyn: Are Your Agent Security Defenses Deployable in Real-World Dynamic Environments?\"", + "authors": "Hao Li, Ruoyao Wen, Shanghao Shi, Ning Zhang, Yevgeniy Vorobeychik, Chaowei Xiao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.03117", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-03", + "updated_at": "2026-05-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-03224-tame-a-trustworthy-test-time-evolution-of-agent-memory-with-systematic-benchmark.md", + "title": "\"TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking\"", + "authors": "Yu Cheng, Yongkang Hu, Jiuan Zhou, Yushuo Zhang, Yihang Chen, Huichi Zhou, Mingang Chen, Zhizhong Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.03224", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-03", + "updated_at": "2026-06-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-03786-aorchestra-automating-sub-agent-creation-for-agentic-orchestration.md", + "title": "\"AOrchestra: Automating Sub-Agent Creation for Agentic Orchestration\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AOrchestra: Automating Sub-Agent Creation for Agentic Orchestration\"", + "authors": "Jianhao Ruan, Zhihao Xu, Yiran Peng, Fashen Ren, Zhaoyang Yu, Xinbing Liang, Jinyu Xiang, Yongru Chen, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.03786", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-03", + "updated_at": "2026-02-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-05115-socialveil-probing-social-intelligence-of-language-agents-under-communication-ba.md", + "title": "\"SocialVeil: Probing Social Intelligence of Language Agents under Communication Barriers\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SocialVeil: Probing Social Intelligence of Language Agents under Communication Barriers\"", + "authors": "Keyang Xuan, Pengda Wang, Chongrui Ye, Haofei Yu, Tal August, Jiaxuan You", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.05115", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-04", + "updated_at": "2026-02-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-05302-piearena-ranking-and-profiling-language-agents-in-realistic-negotiation-scenario.md", + "title": "\"PieArena: Ranking and Profiling Language Agents in Realistic Negotiation Scenarios\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PieArena: Ranking and Profiling Language Agents in Realistic Negotiation Scenarios\"", + "authors": "Chris Zhu, Sasha Cui, Will Sanok Dufallo, Runzhi Jin, Zhen Xu, Linjun Zhang, Daylian Cain", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.05302", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-05", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-05386-spider-sense-intrinsic-risk-sensing-for-efficient-agent-defense-with-hierarchica.md", + "title": "\"Spider-Sense: Intrinsic Risk Sensing for Efficient Agent Defense with Hierarchical Adaptive Screening\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Spider-Sense: Intrinsic Risk Sensing for Efficient Agent Defense with Hierarchical Adaptive Screening\"", + "authors": "Zhenxiong Yu, Zhi Yang, Zhiheng Jin, Shuhe Wang, Heng Zhang, Yanlin Fei, Lingfeng Zeng, Fangqi Lou, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.05386", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-05", + "updated_at": "2026-02-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-07391-naamse-framework-for-evolutionary-security-evaluation-of-agents.md", + "title": "\"NAAMSE: Framework for Evolutionary Security Evaluation of Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"NAAMSE: Framework for Evolutionary Security Evaluation of Agents\"", + "authors": "Kunal Pai, Parth Shah, Harshil Patel", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.07391", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-07", + "updated_at": "2026-03-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-07652-agent-fence-mapping-security-vulnerabilities-across-deep-research-agents.md", + "title": "\"Agent-Fence: Mapping Security Vulnerabilities Across Deep Research Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agent-Fence: Mapping Security Vulnerabilities Across Deep Research Agents\"", + "authors": "Sai Puppala, Ismail Hossain, Md Jahangir Alam, Yoonpyo Lee, Jay Yoo, Tanzim Ahad, Syed Bahauddin Alam, Sajedul Talukder", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.07652", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-07", + "updated_at": "2026-02-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-07962-loca-bench-benchmarking-language-agents-under-controllable-and-extreme-context-g.md", + "title": "\"LOCA-bench: Benchmarking Language Agents Under Controllable and Extreme Context Growth\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LOCA-bench: Benchmarking Language Agents Under Controllable and Extreme Context Growth\"", + "authors": "Weihao Zeng, Yuzhen Huang, Junxian He", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.07962", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-08", + "updated_at": "2026-02-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-08082-spectral-guardrails-for-agents-in-the-wild-detecting-tool-use-hallucinations-via.md", + "title": "\"Spectral Guardrails for Agents in the Wild: Detecting Tool Use Hallucinations via Attention Topology\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Spectral Guardrails for Agents in the Wild: Detecting Tool Use Hallucinations via Attention Topology\"", + "authors": "Valentin Noël", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.08082", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-08", + "updated_at": "2026-02-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "eess.SP" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-08412-from-assistant-to-double-agent-formalizing-and-benchmarking-attacks-on-openclaw-.md", + "title": "\"From Assistant to Double Agent: Formalizing and Benchmarking Attacks on OpenClaw for Personalized Local AI Agent\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Assistant to Double Agent: Formalizing and Benchmarking Attacks on OpenClaw for Personalized Local AI Agent\"", + "authors": "Yuhang Wang, Feiming Xu, Zheng Lin, Guangyu He, Yuzhe Huang, Haichang Gao, Zhenxing Niu, Shiguo Lian, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.08412", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-09", + "updated_at": "2026-02-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-10133-agenttrace-a-structured-logging-framework-for-agent-system-observability.md", + "title": "\"AgentTrace: A Structured Logging Framework for Agent System Observability\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentTrace: A Structured Logging Framework for Agent System Observability\"", + "authors": "Adam AlSayyad, Kelvin Yuxiang Huang, Richik Pal", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.10133", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-07", + "updated_at": "2026-02-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-11749-air-improving-agent-safety-through-incident-response.md", + "title": "\"AIR: Improving Agent Safety through Incident Response\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AIR: Improving Agent Safety through Incident Response\"", + "authors": "Zibo Xiao, Jun Sun, Junjie Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.11749", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-12", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-13379-unsafer-in-many-turns-benchmarking-and-defending-multi-turn-safety-risks-in-tool.md", + "title": "\"Unsafer in Many Turns: Benchmarking and Defending Multi-Turn Safety Risks in Tool-Using Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Unsafer in Many Turns: Benchmarking and Defending Multi-Turn Safety Risks in Tool-Using Agents\"", + "authors": "Xu Li, Simon Yu, Minzhou Pan, Yiyou Sun, Bo Li, Dawn Song, Xue Lin, Weiyan Shi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.13379", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-13", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-13530-remem-reasoning-with-episodic-memory-in-language-agent.md", + "title": "\"REMem: Reasoning with Episodic Memory in Language Agent\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"REMem: Reasoning with Episodic Memory in Language Agent\"", + "authors": "Yiheng Shu, Saisri Padmaja Jonnalagedda, Xiang Gao, Bernal Jiménez Gutiérrez, Weijian Qi, Kamalika Das, Huan Sun, Yu Su", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.13530", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-13", + "updated_at": "2026-02-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-13665-hyfunc-accelerating-llm-based-function-calls-for-agentic-ai-through-hybrid-model.md", + "title": "\"HyFunc: Accelerating LLM-based Function Calls for Agentic AI through Hybrid-Model Cascade and Dynamic Templating\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HyFunc: Accelerating LLM-based Function Calls for Agentic AI through Hybrid-Model Cascade and Dynamic Templating\"", + "authors": "Weibin Liao, Jian-guang Lou, Haoyi Xiong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.13665", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-14", + "updated_at": "2026-02-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-14234-redsearcher-a-scalable-and-cost-efficient-framework-for-long-horizon-search-agen.md", + "title": "\"REDSearcher: A Scalable and Cost-Efficient Framework for Long-Horizon Search Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"REDSearcher: A Scalable and Cost-Efficient Framework for Long-Horizon Search Agents\"", + "authors": "Zheng Chu, Xiao Wang, Jack Hong, Huiming Fan, Yuqi Huang, Yue Yang, Guohai Xu, Chenxiao Zhao, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.14234", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-15", + "updated_at": "2026-02-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-14281-mcpshield-a-security-cognition-layer-for-adaptive-trust-calibration-in-model-con.md", + "title": "\"MCPShield: A Security Cognition Layer for Adaptive Trust Calibration in Model Context Protocol Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MCPShield: A Security Cognition Layer for Adaptive Trust Calibration in Model Context Protocol Agents\"", + "authors": "Zhenhong Zhou, Yuanhe Zhang, Hongwei Cai, Moayad Aloqaily, Ouns Bouachir, Linsey Pang, Prakhar Mehrotra, Kun Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.14281", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-15", + "updated_at": "2026-02-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-16931-narrow-fine-tuning-erodes-safety-alignment-in-vision-language-agents.md", + "title": "Narrow Fine-Tuning Erodes Safety Alignment in Vision-Language Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Narrow Fine-Tuning Erodes Safety Alignment in Vision-Language Agents", + "authors": "Idhant Gulati, Shivam Raval", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.16931", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-18", + "updated_at": "2026-03-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-18456-beyond-single-channel-agentic-benchmarking.md", + "title": "Beyond single-channel agentic benchmarking", + "type": "paper", + "meta": { + "type": "paper", + "title": "Beyond single-channel agentic benchmarking", + "authors": "Nelu D. Radpour", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.18456", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-05", + "updated_at": "2026-02-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY", + "cs.AI", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-19008-capable-but-unreliable-canonical-path-deviation-as-a-causal-mechanism-of-agent-f.md", + "title": "\"Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks\"", + "authors": "Wilson Y. Lee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.19008", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-22", + "updated_at": "2026-02-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-21127-are-you-sure-an-empirical-study-of-human-perception-vulnerability-in-llm-driven-.md", + "title": "\"\\\"Are You Sure?\\\": An Empirical Study of Human Perception Vulnerability in LLM-Driven Agentic Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"\\\"Are You Sure?\\\": An Empirical Study of Human Perception Vulnerability in LLM-Driven Agentic Systems\"", + "authors": "Xinfeng Li, Shenyu Dai, Kelong Zheng, Yue Xiao, Gelei Deng, Wei Dong, Xiaofeng Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.21127", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-24", + "updated_at": "2026-02-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.HC", + "cs.AI", + "cs.CR", + "cs.SI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2602-23320-parammem-augmenting-language-agents-with-parametric-reflective-memory.md", + "title": "\"ParamMem: Augmenting Language Agents with Parametric Reflective Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ParamMem: Augmenting Language Agents with Parametric Reflective Memory\"", + "authors": "Tianjun Yao, Yongqiang Chen, Yujia Zheng, Pan Li, Zhiqiang Shen, Kun Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2602.23320", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-26", + "updated_at": "2026-02-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-00131-thought-virus-viral-misalignment-via-subliminal-prompting-in-multi-agent-systems.md", + "title": "\"Thought Virus: Viral Misalignment via Subliminal Prompting in Multi-Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Thought Virus: Viral Misalignment via Subliminal Prompting in Multi-Agent Systems\"", + "authors": "Moritz Weckbecker, Jonas Müller, Ben Hagag, Michael Mulet", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.00131", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-23", + "updated_at": "2026-02-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-00623-tracesir-a-multi-agent-framework-for-structured-analysis-and-reporting-of-agenti.md", + "title": "\"TraceSIR: A Multi-Agent Framework for Structured Analysis and Reporting of Agentic Execution Traces\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TraceSIR: A Multi-Agent Framework for Structured Analysis and Reporting of Agentic Execution Traces\"", + "authors": "Shu-Xun Yang, Cunxiang Wang, Haoke Zhang, Wenbo Yu, Lindong Wu, Jiayi Gui, Dayong Yang, Yukuo Cen, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.00623", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-28", + "updated_at": "2026-02-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-00801-the-synthetic-web-adversarially-curated-mini-internets-for-diagnosing-epistemic-.md", + "title": "\"The Synthetic Web: Adversarially-Curated Mini-Internets for Diagnosing Epistemic Weaknesses of Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Synthetic Web: Adversarially-Curated Mini-Internets for Diagnosing Epistemic Weaknesses of Language Agents\"", + "authors": "Shrey Shah, Levent Ozgur", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.00801", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-28", + "updated_at": "2026-02-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-01438-enhancing-persona-following-at-decoding-time-via-dynamic-importance-estimation-f.md", + "title": "Enhancing Persona Following at Decoding Time via Dynamic Importance Estimation for Role-Playing Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Enhancing Persona Following at Decoding Time via Dynamic Importance Estimation for Role-Playing Agents", + "authors": "Yuxin Liu, Mingye Zhu, Siyuan Liu, Bo Hu, Lei Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.01438", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-02", + "updated_at": "2026-03-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "planning", + "rag", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-01712-ft-dojo-towards-autonomous-llm-fine-tuning-with-language-agents.md", + "title": "\"FT-Dojo: Towards Autonomous LLM Fine-Tuning with Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"FT-Dojo: Towards Autonomous LLM Fine-Tuning with Language Agents\"", + "authors": "Qizheng Li, Yifei Zhang, Xiao Yang, Xu Yang, Zhuo Wang, Weiqing Liu, Jiang Bian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.01712", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-02", + "updated_at": "2026-05-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-02711-a-natural-language-agentic-approach-to-study-affective-polarization.md", + "title": "A Natural Language Agentic Approach to Study Affective Polarization", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Natural Language Agentic Approach to Study Affective Polarization", + "authors": "Stephanie Anneris Malvicini, Ewelina Gajewska, Arda Derbent, Katarzyna Budzynska, Jarosław A. Chudziak, Maria Vanina Martinez", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.02711", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-03", + "updated_at": "2026-03-03", + "status": "queued", + "relevance": "high", + "topics": [ + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-03515-the-controllability-trap-a-governance-framework-for-military-ai-agents.md", + "title": "\"The Controllability Trap: A Governance Framework for Military AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Controllability Trap: A Governance Framework for Military AI Agents\"", + "authors": "Subramanyam Sahoo", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.03515", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-03", + "updated_at": "2026-03-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-03680-mage-meta-reinforcement-learning-for-language-agents-toward-strategic-exploratio.md", + "title": "\"MAGE: Meta-Reinforcement Learning for Language Agents toward Strategic Exploration and Exploitation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MAGE: Meta-Reinforcement Learning for Language Agents toward Strategic Exploration and Exploitation\"", + "authors": "Lu Yang, Zelai Xu, Minyang Xie, Jiaxuan Gao, Zhao Shok, Yu Wang, Yi Wu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.03680", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-04", + "updated_at": "2026-03-04", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-05553-eigendata-a-self-evolving-multi-agent-platform-for-function-calling-data-synthes.md", + "title": "\"EigenData: A Self-Evolving Multi-Agent Platform for Function-Calling Data Synthesis, Auditing, and Repair\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EigenData: A Self-Evolving Multi-Agent Platform for Function-Calling Data Synthesis, Auditing, and Repair\"", + "authors": "Jiaao Chen, Jingyuan Qi, Mingye Gao, Wei-Chen Wang, Hanrui Wang, Di Jin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.05553", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-05", + "updated_at": "2026-03-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-05578-tool-genesis-a-task-driven-tool-creation-benchmark-for-self-evolving-language-ag.md", + "title": "\"Tool-Genesis: A Task-Driven Tool Creation Benchmark for Self-Evolving Language Agent\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Tool-Genesis: A Task-Driven Tool Creation Benchmark for Self-Evolving Language Agent\"", + "authors": "Bowei Xia, Mengkang Hu, Shijian Wang, Jiarui Jin, Wenxiang Jiao, Yuan Lu, Kexin Li, Ping Luo", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.05578", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-05", + "updated_at": "2026-03-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-07496-from-thinker-to-society-security-in-hierarchical-autonomy-evolution-of-ai-agents.md", + "title": "\"From Thinker to Society: Security in Hierarchical Autonomy Evolution of AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Thinker to Society: Security in Hierarchical Autonomy Evolution of AI Agents\"", + "authors": "Xiaolei Zhang, Lu Zhou, Xiaogang Xu, Jiafei Wu, Tianyu Du, Heqing Huang, Hao Peng, Zhe Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.07496", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-08", + "updated_at": "2026-03-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-07557-agentraft-automated-detection-of-data-over-exposure-in-llm-agents.md", + "title": "\"AgentRaft: Automated Detection of Data Over-Exposure in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentRaft: Automated Detection of Data Over-Exposure in LLM Agents\"", + "authors": "Yixi Lin, Jiangrong Wu, Yuhong Nan, Xueqiang Wang, Xinyuan Zhang, Zibin Zheng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.07557", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-08", + "updated_at": "2026-03-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-07980-onemillion-bench-how-far-are-language-agents-from-human-experts.md", + "title": "\"\\\\$OneMillion-Bench: How Far are Language Agents from Human Experts?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"\\\\$OneMillion-Bench: How Far are Language Agents from Human Experts?\"", + "authors": "Qianyu Yang, Yang Liu, Jiaqi Li, Jun Bai, Hao Chen, Kaiyuan Chen, Tiliang Duan, Jiayun Dong, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.07980", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-09", + "updated_at": "2026-03-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-08721-kernelcraft-benchmarking-for-agentic-close-to-metal-kernel-generation-on-emergin.md", + "title": "\"KernelCraft: Benchmarking for Agentic Close-to-Metal Kernel Generation on Emerging Hardware\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"KernelCraft: Benchmarking for Agentic Close-to-Metal Kernel Generation on Emerging Hardware\"", + "authors": "Jiayi Nie, Haoran Wu, Yao Lai, Zeyu Cao, Cheng Zhang, Binglei Lou, Erwei Wang, Jianyi Cheng, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.08721", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-10", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AR", + "cs.LG", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-09002-security-considerations-for-multi-agent-systems.md", + "title": "Security Considerations for Multi-agent Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Security Considerations for Multi-agent Systems", + "authors": "Tam Nguyen, Moses Ndebugre, Dheeraj Arremsetty", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.09002", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-09", + "updated_at": "2026-04-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-10492-human-ai-co-reasoning-for-clinical-diagnosis-with-evidence-integrated-language-a.md", + "title": "Human-AI Co-reasoning for Clinical Diagnosis with Evidence-Integrated Language Agent", + "type": "paper", + "meta": { + "type": "paper", + "title": "Human-AI Co-reasoning for Clinical Diagnosis with Evidence-Integrated Language Agent", + "authors": "Zhongzhen Huang, Yan Ling, Hong Chen, Ye Feng, Li Wu, Linjie Mu, Shaoting Zhang, Xiaofan Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.10492", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-11", + "updated_at": "2026-03-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-11088-the-attack-and-defense-landscape-of-agentic-ai-a-comprehensive-survey.md", + "title": "\"The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey\"", + "authors": "Juhee Kim, Xiaoyuan Liu, Zhun Wang, Shi Qiu, Bo Li, Wenbo Guo, Dawn Song", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.11088", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-11", + "updated_at": "2026-03-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-11890-quare-quality-aware-requirements-analysis-through-multi-agent-dialectical-negoti.md", + "title": "\"QUARE: Quality-Aware Requirements Analysis through Multi-Agent Dialectical Negotiation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"QUARE: Quality-Aware Requirements Analysis through Multi-Agent Dialectical Negotiation\"", + "authors": "Haowei Cheng, Milhan Kim, Foutse Khomh, Teeradaj Racharak, Nobukazu Yoshioka, Naoyasu Ubayashi, Hironori Washizaki", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.11890", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-12", + "updated_at": "2026-06-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-15309-cctu-a-benchmark-for-tool-use-under-complex-constraints.md", + "title": "\"CCTU: A Benchmark for Tool Use under Complex Constraints\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CCTU: A Benchmark for Tool Use under Complex Constraints\"", + "authors": "Junjie Ye, Guoqiang Zhang, Wenjie Fu, Tao Gui, Qi Zhang, Xuanjing Huang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.15309", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-16", + "updated_at": "2026-03-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-15666-compiled-memory-not-more-information-but-more-precise-instructions-for-language-.md", + "title": "\"Compiled Memory: Not More Information, but More Precise Instructions for Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Compiled Memory: Not More Information, but More Precise Instructions for Language Agents\"", + "authors": "James Rhodes, George Kang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.15666", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-12", + "updated_at": "2026-03-12", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-16734-differential-harm-propensity-in-personalized-llm-agents-the-curious-case-of-ment.md", + "title": "\"Differential Harm Propensity in Personalized LLM Agents: The Curious Case of Mental Health Disclosure\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Differential Harm Propensity in Personalized LLM Agents: The Curious Case of Mental Health Disclosure\"", + "authors": "Caglar Yildirim", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.16734", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-17", + "updated_at": "2026-03-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-17392-agentic-cognitive-profiling-realigning-automated-alzheimer-s-disease-detection-w.md", + "title": "\"Agentic Cognitive Profiling: Realigning Automated Alzheimer's Disease Detection with Clinical Construct Validity\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic Cognitive Profiling: Realigning Automated Alzheimer's Disease Detection with Clinical Construct Validity\"", + "authors": "Jiawen Kang, Kun Li, Dongrui Han, Jinchao Li, Junan Li, Lingwei Meng, Xixin Wu, Helen Meng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.17392", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-18", + "updated_at": "2026-03-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.IR", + "q-bio.NC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-18245-who-tests-the-testers-systematic-enumeration-and-coverage-audit-of-llm-agent-too.md", + "title": "Who Tests the Testers? Systematic Enumeration and Coverage Audit of LLM Agent Tool Call Safety", + "type": "paper", + "meta": { + "type": "paper", + "title": "Who Tests the Testers? Systematic Enumeration and Coverage Audit of LLM Agent Tool Call Safety", + "authors": "Xuan Chen, Lu Yan, Ruqi Zhang, Xiangyu Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.18245", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-18", + "updated_at": "2026-03-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-19469-a-framework-for-formalizing-llm-agent-security.md", + "title": "A Framework for Formalizing LLM Agent Security", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Framework for Formalizing LLM Agent Security", + "authors": "Vincent Siu, Jingxuan He, Kyle Montgomery, Zhun Wang, Neil Gong, Chenguang Wang, Dawn Song", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.19469", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-19", + "updated_at": "2026-03-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-19684-tsegagent-zero-shot-tooth-segmentation-via-geometry-aware-vision-language-agents.md", + "title": "\"TSegAgent: Zero-Shot Tooth Segmentation via Geometry-Aware Vision-Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TSegAgent: Zero-Shot Tooth Segmentation via Geometry-Aware Vision-Language Agents\"", + "authors": "Shaojie Zhuang, Lu Yin, Guangshun Wei, Yunpeng Li, Xilu Wang, Yuanfeng Zhou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.19684", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-20", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-21357-agenther-hindsight-experience-replay-for-llm-agent-trajectory-relabeling.md", + "title": "\"AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling\"", + "authors": "Liang Ding", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.21357", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-22", + "updated_at": "2026-05-10", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "memory", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-21564-toward-a-theory-of-hierarchical-memory-for-language-agents.md", + "title": "Toward a Theory of Hierarchical Memory for Language Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Toward a Theory of Hierarchical Memory for Language Agents", + "authors": "Yashar Talebirad, Ali Parsaee, Csongor Y. Szepesvari, Amirhossein Nadiri, Osmar Zaiane", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.21564", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-23", + "updated_at": "2026-03-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI", + "cs.IT", + "cs.SI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-24257-memory-augmented-vision-language-agents-for-persistent-and-semantically-consiste.md", + "title": "Memory-Augmented Vision-Language Agents for Persistent and Semantically Consistent Object Captioning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Memory-Augmented Vision-Language Agents for Persistent and Semantically Consistent Object Captioning", + "authors": "Tommaso Galliena, Stefano Rosa, Tommaso Apicella, Pietro Morerio, Alessio Del Bue, Lorenzo Natale", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.24257", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-25", + "updated_at": "2026-03-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-25353-safeguard-asf-sr-agentic-humanoid-robot-system-for-autonomous-industrial-safety.md", + "title": "\"SafeGuard ASF: SR Agentic Humanoid Robot System for Autonomous Industrial Safety\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SafeGuard ASF: SR Agentic Humanoid Robot System for Autonomous Industrial Safety\"", + "authors": "Thanh Nguyen Canh, Thang Tran Viet, Thanh Tuan Tran, Ben Wei Lim", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.25353", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-26", + "updated_at": "2026-03-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "embodied-agent", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-27148-safetydrift-predicting-when-ai-agents-cross-the-line-before-they-actually-do.md", + "title": "\"SafetyDrift: Predicting When AI Agents Cross the Line Before They Actually Do\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SafetyDrift: Predicting When AI Agents Cross the Line Before They Actually Do\"", + "authors": "Aditya Dhodapkar, Farhaan Pishori", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.27148", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-28", + "updated_at": "2026-03-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-28166-evaluating-privilege-usage-of-agents-with-real-world-tools.md", + "title": "Evaluating Privilege Usage of Agents with Real-World Tools", + "type": "paper", + "meta": { + "type": "paper", + "title": "Evaluating Privilege Usage of Agents with Real-World Tools", + "authors": "Quan Zhang, Lianhang Fu, Lvsi Lian, Gwihwan Go, Yujue Wang, Chijin Zhou, Yu Jiang, Geguang Pu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.28166", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-30", + "updated_at": "2026-04-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-28428-synergy-a-next-generation-general-purpose-agent-for-open-agentic-web.md", + "title": "\"Synergy: A Next-Generation General-Purpose Agent for Open Agentic Web\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Synergy: A Next-Generation General-Purpose Agent for Open Agentic Web\"", + "authors": "Xiaohang Nie, Zihan Guo, Kezhuo Yang, Zhichong Zheng, Bochen Ge, Shuai Pan, Zeyi Chen, Youling Xiang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.28428", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-30", + "updated_at": "2026-03-30", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "embodied-agent", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2603-28900-robust-multi-agent-reinforcement-learning-for-small-uas-separation-assurance-und.md", + "title": "Robust Multi-Agent Reinforcement Learning for Small UAS Separation Assurance under GPS Degradation and Spoofing", + "type": "paper", + "meta": { + "type": "paper", + "title": "Robust Multi-Agent Reinforcement Learning for Small UAS Separation Assurance under GPS Degradation and Spoofing", + "authors": "Alex Zongo, Filippos Fotiadis, Ufuk Topcu, Peng Wei", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2603.28900", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-03-30", + "updated_at": "2026-03-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO", + "cs.AI", + "cs.LG", + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-02022-atbench-a-diverse-and-realistic-agent-trajectory-benchmark-for-safety-evaluation.md", + "title": "\"ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis\"", + "authors": "Yu Li, Haoyu Luo, Yuejin Xie, Yuqian Fu, Zhonghao Yang, Shuai Shao, Qihan Ren, Wanying Qu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.02022", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-02", + "updated_at": "2026-05-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-02155-brief-is-better-non-monotonic-chain-of-thought-budget-effects-in-function-callin.md", + "title": "\"Brief Is Better: Non-Monotonic Chain-of-Thought Budget Effects in Function-Calling Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Brief Is Better: Non-Monotonic Chain-of-Thought Budget Effects in Function-Calling Language Agents\"", + "authors": "Xuan Qi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.02155", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-02", + "updated_at": "2026-04-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "function-calling, language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-03098-co-evolution-of-policy-and-internal-reward-for-language-agents.md", + "title": "Co-Evolution of Policy and Internal Reward for Language Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Co-Evolution of Policy and Internal Reward for Language Agents", + "authors": "Xinyu Wang, Hanwei Wu, Jingwei Song, Shuyuan Zhang, Jiayi Zhang, Fanqi Kong, Tung Sum Thomas Kwok, Xiao-Wen Chang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.03098", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-03", + "updated_at": "2026-04-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-03242-draft-task-decoupled-latent-reasoning-for-agent-safety.md", + "title": "\"DRAFT: Task Decoupled Latent Reasoning for Agent Safety\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DRAFT: Task Decoupled Latent Reasoning for Agent Safety\"", + "authors": "Lin Wang, Junfeng Fang, Dan Zhang, Fei Shen, Xiang Wang, Tat-Seng Chua", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.03242", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-02-11", + "updated_at": "2026-02-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-04131-profile-then-reason-bounded-semantic-complexity-for-tool-augmented-language-agen.md", + "title": "\"Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents\"", + "authors": "Paulo Akira F. Enabe", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.04131", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-05", + "updated_at": "2026-04-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-04426-shieldnet-network-level-guardrails-against-emerging-supply-chain-injections-in-a.md", + "title": "\"ShieldNet: Network-Level Guardrails against Emerging Supply-Chain Injections in Agentic Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ShieldNet: Network-Level Guardrails against Emerging Supply-Chain Injections in Agentic Systems\"", + "authors": "Zhuowen Yuan, Zhaorun Chen, Zhen Xiang, Nathaniel D. Bastian, Seyyed Hadi Hashemi, Chaowei Xiao, Wenbo Guo, Bo Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.04426", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-06", + "updated_at": "2026-04-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-06762-arulecon-agentic-security-rule-conversion.md", + "title": "\"ARuleCon: Agentic Security Rule Conversion\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ARuleCon: Agentic Security Rule Conversion\"", + "authors": "Ming Xu, Hongtai Wang, Yanpei Guo, Zhengmin Yu, Weili Han, Hoon Wei Lim, Jin Song Dong, Jiaheng Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.06762", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-08", + "updated_at": "2026-04-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-06972-differentiable-environment-trajectory-co-optimization-for-safe-multi-agent-navig.md", + "title": "Differentiable Environment-Trajectory Co-Optimization for Safe Multi-Agent Navigation", + "type": "paper", + "meta": { + "type": "paper", + "title": "Differentiable Environment-Trajectory Co-Optimization for Safe Multi-Agent Navigation", + "authors": "Zhan Gao, Gabriele Fadini, Stelian Coros, Amanda Prorok", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.06972", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-08", + "updated_at": "2026-04-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-08388-awakening-the-sleeping-agent-lean-specific-agentic-data-reactivates-general-tool.md", + "title": "\"Awakening the Sleeping Agent: Lean-Specific Agentic Data Reactivates General Tool Use in Goedel Prover\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Awakening the Sleeping Agent: Lean-Specific Agentic Data Reactivates General Tool Use in Goedel Prover\"", + "authors": "Jui-Hui Chung, Hongzhou Lin, Lai Jiang, Shange Tang, Chi Jin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.08388", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-09", + "updated_at": "2026-04-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-10577-the-blind-spot-of-agent-safety-how-benign-user-instructions-expose-critical-vuln.md", + "title": "\"The Blind Spot of Agent Safety: How Benign User Instructions Expose Critical Vulnerabilities in Computer-Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Blind Spot of Agent Safety: How Benign User Instructions Expose Critical Vulnerabilities in Computer-Use Agents\"", + "authors": "Xuwei Ding, Skylar Zhai, Linxin Song, Jiate Li, Taiwei Shi, Nicholas Meade, Siva Reddy, Jian Kang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.10577", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-12", + "updated_at": "2026-04-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-11557-unitoolcall-unifying-tool-use-representation-data-and-evaluation-for-llm-agents.md", + "title": "\"UniToolCall: Unifying Tool-Use Representation, Data, and Evaluation for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"UniToolCall: Unifying Tool-Use Representation, Data, and Evaluation for LLM Agents\"", + "authors": "Yijuan Liang, Xinghao Chen, Yifan Ge, Ziyi Wu, Hao Wu, Changyu Zeng, Wei Xing, Xiaoyu Shen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.11557", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-13", + "updated_at": "2026-05-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-12986-parallax-why-ai-agents-that-think-must-never-act.md", + "title": "\"Parallax: Why AI Agents That Think Must Never Act\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Parallax: Why AI Agents That Think Must Never Act\"", + "authors": "Joel Fokou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.12986", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-14", + "updated_at": "2026-04-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-13298-can-agents-secure-hardware-evaluating-agentic-llm-driven-obfuscation-for-ip-prot.md", + "title": "Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection", + "authors": "Sujan Ghimire, Parsa Mirfasihi, Muhtasim Alam Chowdhury, Veeramani Pugazhenthi, Harish Kumar Dharavath, Farshad Firouzi, Rozhin Yasaei, Pratik Satam, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.13298", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-14", + "updated_at": "2026-04-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-13536-don-t-let-ai-agents-yolo-your-files-shifting-information-and-control-to-filesyst.md", + "title": "\"Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy\"", + "authors": "Shawn Wanxiang Zhong, Junxuan Liao, Jing Liu, Mai Zheng, Andrea C. Arpaci-Dusseau, Remzi H. Arpaci-Dusseau", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.13536", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-15", + "updated_at": "2026-04-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.OS" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-13954-hintbench-horizon-agent-intrinsic-non-attack-trajectory-benchmark.md", + "title": "\"HINTBench: Horizon-agent Intrinsic Non-attack Trajectory Benchmark\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HINTBench: Horizon-agent Intrinsic Non-attack Trajectory Benchmark\"", + "authors": "Jiacheng Wang, Jinchang Hou, Fabian Wang, Ping Jian, Chenfu Bao, Zhonghou Lv", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.13954", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-15", + "updated_at": "2026-04-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-14399-spacemind-a-modular-and-self-evolving-embodied-vision-language-agent-framework-f.md", + "title": "\"SpaceMind: A Modular and Self-Evolving Embodied Vision-Language Agent Framework for Autonomous On-orbit Servicing\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SpaceMind: A Modular and Self-Evolving Embodied Vision-Language Agent Framework for Autonomous On-orbit Servicing\"", + "authors": "Aodi Wu, Haodong Han, Xubo Luo, Ruisuo Wang, Shan He, Xue Wan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.14399", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-15", + "updated_at": "2026-04-15", + "status": "queued", + "relevance": "high", + "topics": [ + "embodied-agent", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO", + "cs.AI", + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-15415-harmfulskillbench-how-do-harmful-skills-weaponize-your-agents.md", + "title": "\"HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?\"", + "authors": "Yukun Jiang, Yage Zhang, Michael Backes, Xinyue Shen, Yang Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.15415", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-16", + "updated_at": "2026-04-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-15579-don-t-make-models-guess-security-and-safety-symbolic-guardrails-for-domain-speci.md", + "title": "\"Don't Make Models Guess Security and Safety: Symbolic Guardrails for Domain-Specific AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Don't Make Models Guess Security and Safety: Symbolic Guardrails for Domain-Specific AI Agents\"", + "authors": "Yining Hong, Yining She, Eunsuk Kang, Christopher S. Timperley, Christian Kästner", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.15579", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-16", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-16706-evaluating-tool-using-language-agents-judge-reliability-propagation-cascades-and.md", + "title": "\"Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench\"", + "authors": "Bhaskar Gurram", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.16706", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-17", + "updated_at": "2026-04-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-17562-safeagent-a-runtime-protection-architecture-for-agentic-systems.md", + "title": "\"SafeAgent: A Runtime Protection Architecture for Agentic Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SafeAgent: A Runtime Protection Architecture for Agentic Systems\"", + "authors": "Hailin Liu, Eugene Ilyushin, Jie Ni, Min Zhu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.17562", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-19", + "updated_at": "2026-04-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-18658-owner-harm-a-missing-threat-model-for-ai-agent-safety.md", + "title": "\"Owner-Harm: A Missing Threat Model for AI Agent Safety\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Owner-Harm: A Missing Threat Model for AI Agent Safety\"", + "authors": "Dongcheng Zhang, Yiqing Jiang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.18658", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-20", + "updated_at": "2026-04-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-18718-towards-optimal-agentic-architectures-for-offensive-security-tasks.md", + "title": "Towards Optimal Agentic Architectures for Offensive Security Tasks", + "type": "paper", + "meta": { + "type": "paper", + "title": "Towards Optimal Agentic Architectures for Offensive Security Tasks", + "authors": "Isaac David, Arthur Gervais", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.18718", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-20", + "updated_at": "2026-04-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-18847-human-guided-harm-recovery-for-computer-use-agents.md", + "title": "Human-Guided Harm Recovery for Computer Use Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Human-Guided Harm Recovery for Computer Use Agents", + "authors": "Christy Li, Sky CH-Wang, Andi Peng, Andreea Bobu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.18847", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-20", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-19821-jtpro-a-joint-tool-prompt-reflective-optimization-framework-for-language-agents.md", + "title": "\"JTPRO: A Joint Tool-Prompt Reflective Optimization Framework for Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"JTPRO: A Joint Tool-Prompt Reflective Optimization Framework for Language Agents\"", + "authors": "Sandip Ghoshal, Anshul Mittal, Jyotika Singh, Miguel Ballesteros, Weiyi Sun, Fang Tu, Shailender Singh, Yassine Benajiba, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.19821", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-20", + "updated_at": "2026-04-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-19844-if-you-re-waiting-for-a-sign-that-might-not-be-it-mitigating-trust-boundary-conf.md", + "title": "\"If you're waiting for a sign... that might not be it! Mitigating Trust Boundary Confusion from Visual Injections on Vision-Language Agentic Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"If you're waiting for a sign... that might not be it! Mitigating Trust Boundary Confusion from Visual Injections on Vision-Language Agentic Systems\"", + "authors": "Jiamin Chang, Minhui Xue, Ruoxi Sun, Shuchao Pang, Salil S. Kanhere, Hammond Pearce", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.19844", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-21", + "updated_at": "2026-04-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-20994-breaking-mcp-with-function-hijacking-attacks-novel-threats-for-function-calling-.md", + "title": "\"Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models\"", + "authors": "Yannis Belkhiter, Giulio Zizzo, Sergio Maffeis, Seshu Tirupathi, John D. Kelleher", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.20994", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-22", + "updated_at": "2026-04-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-21190-spatio-adaptive-test-time-orchestration-of-vision-language-agents-for-spatial-re.md", + "title": "\"SpatiO: Adaptive Test-Time Orchestration of Vision-Language Agents for Spatial Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SpatiO: Adaptive Test-Time Orchestration of Vision-Language Agents for Spatial Reasoning\"", + "authors": "Chan Yeong Hwang, Miso Choi, Sunghyun On, Jinkyu Kim, Jungbeom Lee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.21190", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-23", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-22879-beyond-single-agent-alignment-preventing-context-fragmented-violations-in-multi-.md", + "title": "\"Beyond Single-Agent Alignment: Preventing Context-Fragmented Violations in Multi-Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Single-Agent Alignment: Preventing Context-Fragmented Violations in Multi-Agent Systems\"", + "authors": "Jie Wu, Ming Gong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.22879", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-24", + "updated_at": "2026-04-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.AI", + "cs.CR", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-23210-discovering-agentic-safety-specifications-from-1-bit-danger-signals.md", + "title": "Discovering Agentic Safety Specifications from 1-Bit Danger Signals", + "type": "paper", + "meta": { + "type": "paper", + "title": "Discovering Agentic Safety Specifications from 1-Bit Danger Signals", + "authors": "Víctor Gallego", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.23210", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-25", + "updated_at": "2026-04-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-23374-ghost-in-the-agent-redefining-information-flow-tracking-for-llm-agents.md", + "title": "\"Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents\"", + "authors": "Yuandao Cai, Wensheng Tang, Cheng Wen, Shengchao Qin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.23374", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-25", + "updated_at": "2026-04-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-23459-architecture-matters-for-multi-agent-security.md", + "title": "Architecture Matters for Multi-Agent Security", + "type": "paper", + "meta": { + "type": "paper", + "title": "Architecture Matters for Multi-Agent Security", + "authors": "Ben Hagag, William L. Anderson, Christian Schroeder de Witt, Sarah Scheffler", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.23459", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-25", + "updated_at": "2026-04-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "planning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.CR", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-24212-empowering-autonomous-debugging-agents-with-efficient-dynamic-analysis.md", + "title": "Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis", + "type": "paper", + "meta": { + "type": "paper", + "title": "Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis", + "authors": "Jiahong Xiang, Xiaoyang Xu, Xiaopan Chu, Hongliang Tian, Yuqun Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.24212", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-27", + "updated_at": "2026-04-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-24826-a-comparative-evaluation-of-ai-agent-security-guardrails.md", + "title": "A Comparative Evaluation of AI Agent Security Guardrails", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Comparative Evaluation of AI Agent Security Guardrails", + "authors": "Qi Li, Jiu Li, Pingtao Wei, Jianjun Xu, Xueyi Wei, Jiwei Shi, Xuan Zhang, Yanhui Yang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.24826", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-27", + "updated_at": "2026-04-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-25135-fama-failure-aware-meta-agentic-framework-for-open-source-llms-in-interactive-to.md", + "title": "\"FAMA: Failure-Aware Meta-Agentic Framework for Open-Source LLMs in Interactive Tool Use Environments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"FAMA: Failure-Aware Meta-Agentic Framework for Open-Source LLMs in Interactive Tool Use Environments\"", + "authors": "Amir Saeidi, Venkatesh Mishra, Souradeep Mukhopadhyay, Gaowen Liu, Ali Payani, Jayanth Srinivasa, Chitta Baral", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.25135", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-28", + "updated_at": "2026-04-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-25318-cutscene-agent-an-llm-agent-framework-for-automated-3d-cutscene-generation.md", + "title": "\"Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation\"", + "authors": "Lanshan He, Haozhou Pang, Qi Gan, Xin Shen, Ziwei Zhang, Yibo Liu, Gang Fang, Bo Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.25318", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-28", + "updated_at": "2026-04-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.GR", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-25555-from-crud-to-autonomous-agents-formal-validation-and-zero-trust-security-for-sem.md", + "title": "\"From CRUD to Autonomous Agents: Formal Validation and Zero-Trust Security for Semantic Gateways in AI-Native Enterprise Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From CRUD to Autonomous Agents: Formal Validation and Zero-Trust Security for Semantic Gateways in AI-Native Enterprise Systems\"", + "authors": "Ignacio Peyrano", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.25555", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-28", + "updated_at": "2026-04-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-26274-enforcing-benign-trajectories-a-behavioral-firewall-for-structured-workflow-ai-a.md", + "title": "\"Enforcing Benign Trajectories: A Behavioral Firewall for Structured-Workflow AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Enforcing Benign Trajectories: A Behavioral Firewall for Structured-Workflow AI Agents\"", + "authors": "Hung Dang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.26274", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-29", + "updated_at": "2026-04-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-26959-careguardai-context-aware-multi-agent-guardrails-for-clinical-safety-hallucinati.md", + "title": "\"CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs\"", + "authors": "Elham Nasarian, Abhilash Neog, Kwok-Leung Tsui, Niyousha HosseiniChimeh", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.26959", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-07", + "updated_at": "2026-04-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-27092-end-to-end-autonomous-scientific-discovery-on-a-real-optical-platform.md", + "title": "End-to-end autonomous scientific discovery on a real optical platform", + "type": "paper", + "meta": { + "type": "paper", + "title": "End-to-end autonomous scientific discovery on a real optical platform", + "authors": "Shuxing Yang, Fujia Chen, Rui Zhao, Junyao Wu, Yize Wang, Haiyao Luo, Ning Han, Qiaolu Chen, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.27092", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-29", + "updated_at": "2026-04-29", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "physics.optics" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-27464-security-attack-and-defense-strategies-for-autonomous-agent-frameworks-a-layered.md", + "title": "\"Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study\"", + "authors": "Luyao Xu, Xiang Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.27464", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-30", + "updated_at": "2026-04-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety, autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-27699-bridging-values-and-behavior-a-hierarchical-framework-for-proactive-embodied-age.md", + "title": "\"Bridging Values and Behavior: A Hierarchical Framework for Proactive Embodied Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Bridging Values and Behavior: A Hierarchical Framework for Proactive Embodied Agents\"", + "authors": "Chunhui Zhang, Yuxuan Wang, Aoyang Qin, Yi-Long Lu, Kunlun Wu, Yizhou Wang, Wei Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.27699", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-30", + "updated_at": "2026-04-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-27859-rethinking-agentic-reinforcement-learning-in-large-language-models.md", + "title": "Rethinking Agentic Reinforcement Learning In Large Language Models", + "type": "paper", + "meta": { + "type": "paper", + "title": "Rethinking Agentic Reinforcement Learning In Large Language Models", + "authors": "Fangming Cui, Ruixiao Zhu, Cheng Fang, Sunan Li, Jiahong Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.27859", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-30", + "updated_at": "2026-05-15", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.ET" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2604-28157-flashrt-towards-computationally-and-memory-efficient-red-teaming-for-prompt-inje.md", + "title": "\"FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption\"", + "authors": "Yanting Wang, Chenlong Yin, Ying Chen, Jinyuan Jia", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2604.28157", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-30", + "updated_at": "2026-04-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-00081-alignment-contracts-for-agentic-security-systems.md", + "title": "Alignment Contracts for Agentic Security Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Alignment Contracts for Agentic Security Systems", + "authors": "Isaac David, Marco Guarnieri, Arthur Gervais", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.00081", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-30", + "updated_at": "2026-04-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.LO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-00741-self-adaptive-multi-agent-llm-based-security-pattern-selection-for-iot-systems.md", + "title": "Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems", + "authors": "Saeid Jamshidi, Foutse Khomh, Carol Fung, Kawser Wazed Nafi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.00741", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-01", + "updated_at": "2026-05-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-00845-graph-query-generation-with-constraint-guided-large-language-agents.md", + "title": "Graph Query Generation with Constraint-guided Large Language Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Graph Query Generation with Constraint-guided Large Language Agents", + "authors": "Mengying Wang, Nicolaas Jedema, Rahul Pandey, RaviKiran Krishnan, Jens Lehmann, Yinghui Wu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.00845", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-09", + "updated_at": "2026-04-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DB", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-01101-virtual-speech-therapist-a-clinician-in-the-loop-ai-speech-therapy-agent-for-per.md", + "title": "\"Virtual Speech Therapist: A Clinician-in-the-Loop AI Speech Therapy Agent for Personalized and Supervised Therapy\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Virtual Speech Therapist: A Clinician-in-the-Loop AI Speech Therapy Agent for Personalized and Supervised Therapy\"", + "authors": "Shakeel Sheikh, Patrick Marmaroli, MD Sahidullah, Slim Ouni, Fabrice Hirsch, Goncalo Leal, Bjorn W Schuller", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.01101", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-01", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.SD", + "eess.AS" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-01644-toward-a-principled-framework-for-agent-safety-measurement.md", + "title": "Toward a Principled Framework for Agent Safety Measurement", + "type": "paper", + "meta": { + "type": "paper", + "title": "Toward a Principled Framework for Agent Safety Measurement", + "authors": "Shuyi Lin, Anshuman Suri, Alina Oprea, Cheng Tan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.01644", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-02", + "updated_at": "2026-05-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-02240-physicianbench-evaluating-llm-agents-in-real-world-ehr-environments.md", + "title": "\"PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments\"", + "authors": "Ruoqi Liu, Imran Q. Mohiuddin, Austin J. Schoeffler, Kavita Renduchintala, Ashwin Nayak, Prasantha L. Vemu, Shivam C. Vedak, Kameron C. Black, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.02240", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-04", + "updated_at": "2026-05-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-03242-enhancing-agent-safety-judgment-controlled-benchmark-rewriting-and-analogical-re.md", + "title": "\"Enhancing Agent Safety Judgment: Controlled Benchmark Rewriting and Analogical Reasoning for Deceptive Out-of-Distribution Scenarios\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Enhancing Agent Safety Judgment: Controlled Benchmark Rewriting and Analogical Reasoning for Deceptive Out-of-Distribution Scenarios\"", + "authors": "Zuoyu Zhang, Yancheng Zhu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.03242", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-05", + "updated_at": "2026-05-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-03312-memflow-intent-driven-memory-orchestration-for-small-language-model-agents.md", + "title": "\"MemFlow: Intent-Driven Memory Orchestration for Small Language Model Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemFlow: Intent-Driven Memory Orchestration for Small Language Model Agents\"", + "authors": "Jiayi Chen, Yingcong Li, Guiling Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.03312", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-05", + "updated_at": "2026-05-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-03328-llm-adam-a-generalizable-llm-agent-framework-for-pre-print-anomaly-detection-in-.md", + "title": "\"LLM-ADAM: A Generalizable LLM Agent Framework for Pre-Print Anomaly Detection in Additive Manufacturing\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LLM-ADAM: A Generalizable LLM Agent Framework for Pre-Print Anomaly Detection in Additive Manufacturing\"", + "authors": "Ahmadreza Eslaminia, Chuhan Cai, Cameron Smith, Ruo-Syuan Mei, Shichen Li, Rajiv Malhotra, Klara Nahrstedt, Chenhui Shao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.03328", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-05", + "updated_at": "2026-05-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-03505-lats-rca-language-agent-tree-search-for-root-cause-analysis-in-microservices.md", + "title": "\"LATS-RCA: Language Agent Tree Search for Root Cause Analysis in Microservices\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LATS-RCA: Language Agent Tree Search for Root Cause Analysis in Microservices\"", + "authors": "Alexander Naakka, Yuqing Wang, Mika V Mäntylä", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.03505", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-05", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-04107-tscg-deterministic-tool-schema-compilation-for-agentic-llm-deployments.md", + "title": "\"TSCG: Deterministic Tool-Schema Compilation for Agentic LLM Deployments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TSCG: Deterministic Tool-Schema Compilation for Agentic LLM Deployments\"", + "authors": "Furkan Sakizli", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.04107", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-04", + "updated_at": "2026-05-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-04808-decodingtrust-agent-platform-dtap-a-controllable-and-interactive-red-teaming-pla.md", + "title": "\"DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents\"", + "authors": "Zhaorun Chen, Xun Liu, Haibo Tong, Chengquan Guo, Yuzhou Nie, Jiawei Zhang, Mintong Kang, Chejian Xu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.04808", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-06", + "updated_at": "2026-05-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-05242-beyond-semantic-similarity-rethinking-retrieval-for-agentic-search-via-direct-co.md", + "title": "\"Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction\"", + "authors": "Zhuofeng Li, Haoxiang Zhang, Cong Wei, Pan Lu, Ping Nie, Yi Lu, Yuyang Bai, Shangbin Feng, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.05242", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-03", + "updated_at": "2026-05-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-05704-safeharbor-hierarchical-memory-augmented-guardrail-for-llm-agent-safety.md", + "title": "\"SafeHarbor: Hierarchical Memory-Augmented Guardrail for LLM Agent Safety\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SafeHarbor: Hierarchical Memory-Augmented Guardrail for LLM Agent Safety\"", + "authors": "Zhe Liu, Zonghao Ying, Wenxin Zhang, Quanchen Zou, Deyue Zhang, Dongdong Yang, Xiangzheng Zhang, Hao Peng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.05704", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "memory", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "agent-safety, autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-05716-more-is-not-always-better-cross-component-interference-in-llm-agent-scaffolding.md", + "title": "\"More Is Not Always Better: Cross-Component Interference in LLM Agent Scaffolding\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"More Is Not Always Better: Cross-Component Interference in LLM Agent Scaffolding\"", + "authors": "Ming Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.05716", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-05-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-06078-milestone-guided-policy-learning-for-long-horizon-language-agents.md", + "title": "Milestone-Guided Policy Learning for Long-Horizon Language Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Milestone-Guided Policy Learning for Long-Horizon Language Agents", + "authors": "Zixuan Wang, Yuchen Yan, Hongxing Li, Teng Pan, Dingming Li, Ruiqing Zhang, Weiming Lu, Jun Xiao, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.06078", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-05-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-06713-agentic-ai-and-the-industrialization-of-cyber-offense-forecast-consequences-and-.md", + "title": "\"Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand\"", + "authors": "Christopher Koch", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.06713", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-06", + "updated_at": "2026-05-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-safety, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-06716-from-storage-to-experience-a-survey-on-the-evolution-of-llm-agent-memory-mechani.md", + "title": "\"From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms\"", + "authors": "Jinghao Luo, Yuchen Tian, Chuxue Cao, Ziyang Luo, Hongzhan Lin, Kaixin Li, Chuyi Kong, Ruichao Yang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.06716", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-05-07", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-06737-a-self-healing-framework-for-reliable-llm-based-autonomous-agents.md", + "title": "A Self-Healing Framework for Reliable LLM-Based Autonomous Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Self-Healing Framework for Reliable LLM-Based Autonomous Agents", + "authors": "Cheonsu Jeong, Younggun Shin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.06737", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-05-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-06812-towards-security-auditable-llm-agents-a-unified-graph-representation.md", + "title": "\"Towards Security-Auditable LLM Agents: A Unified Graph Representation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Towards Security-Auditable LLM Agents: A Unified Graph Representation\"", + "authors": "Chaofan Li, Lyuye Zhang, Jintao Zhai, Siyue Feng, Xichun Yang, Huahao Wang, Shihan Dou, Yu Ji, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.06812", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-05-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-06869-agentick-a-unified-benchmark-for-general-sequential-decision-making-agents.md", + "title": "\"Agentick: A Unified Benchmark for General Sequential Decision-Making Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentick: A Unified Benchmark for General Sequential Decision-Making Agents\"", + "authors": "Roger Creus Castanyer, Pablo Samuel Castro, Glen Berseth", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.06869", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-05-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "23", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-06890-beyond-the-black-box-interpretability-of-agentic-ai-tool-use.md", + "title": "\"Beyond the Black Box: Interpretability of Agentic AI Tool Use\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond the Black Box: Interpretability of Agentic AI Tool Use\"", + "authors": "Hariom Tatsat, Ariye Shater", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.06890", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-06957-learning-and-reusing-policy-decompositions-for-hierarchical-generalized-planning.md", + "title": "Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents", + "authors": "Shirin Sohrabi, Haritha Ananthakrishnan, Harsha Kokel, Kavitha Srinivas, Michael Katz", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.06957", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-05-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-06992-why-does-agentic-safety-fail-to-generalize-across-tasks.md", + "title": "Why Does Agentic Safety Fail to Generalize Across Tasks?", + "type": "paper", + "meta": { + "type": "paper", + "title": "Why Does Agentic Safety Fail to Generalize Across Tasks?", + "authors": "Yonatan Slutzky, Yotam Alexander, Tomer Slor, Yoav Nagel, Nadav Cohen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.06992", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-07", + "updated_at": "2026-05-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "embodied-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "stat.ML" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-07112-switchcraft-ai-model-router-for-agentic-tool-calling.md", + "title": "\"Switchcraft: AI Model Router for Agentic Tool Calling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Switchcraft: AI Model Router for Agentic Tool Calling\"", + "authors": "Sharad Agarwal, Pooria Namyar, Alec Wolman, Rahul Ambavat, Ankur Gupta, Qizheng Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.07112", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-08", + "updated_at": "2026-05-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-07251-can-agents-price-a-reaction-evaluating-llms-on-chemical-cost-reasoning.md", + "title": "Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning", + "authors": "Yuyang Wu, Yue Huang, Shuaike Shen, Xujian Wang, Shuhao Zhang, Qiyao Xue, Weichen Liu, Runtian Gao, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.07251", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-08", + "updated_at": "2026-05-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-07830-cybiasbench-benchmarking-bias-in-llm-agents-for-cyber-attack-scenarios.md", + "title": "\"CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios\"", + "authors": "Taein Lim, Seongyong Ju, Munhyeok Kim, Hyunjun Kim, Hoki Kim", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.07830", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-08", + "updated_at": "2026-05-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-08374-memq-integrating-q-learning-into-self-evolving-memory-agents-over-provenance-dag.md", + "title": "\"MemQ: Integrating Q-Learning into Self-Evolving Memory Agents over Provenance DAGs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemQ: Integrating Q-Learning into Self-Evolving Memory Agents over Provenance DAGs\"", + "authors": "Junwei Liao, Haoting Shi, Ruiwen Zhou, Jiaqian Wang, Shengtao Zhang, Wei Zhang, Ying Wen, Zhiyu Li, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.08374", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-08", + "updated_at": "2026-05-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-08442-defense-effectiveness-across-architectural-layers-a-mechanistic-evaluation-of-pe.md", + "title": "\"Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents\"", + "authors": "Jun Wen Leong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.08442", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-08", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "23", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-08763-when-llms-team-up-a-coordinated-attack-framework-for-automated-cyber-intrusions.md", + "title": "\"When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions\"", + "authors": "Minfeng Qi, Tianqing Zhu, Zijie Xu, Congcong Zhu, Qin Wang, Wanlei Zhou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.08763", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-09", + "updated_at": "2026-05-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-08876-otora-a-unified-red-teaming-framework-for-reasoning-level-denial-of-service-in-l.md", + "title": "\"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents\"", + "authors": "Xinyu Li, Ronghui Mu, Lin Li, Tianjin Huang, Gaojie Jin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.08876", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-09", + "updated_at": "2026-06-07", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-08964-trustworthy-ai-ensuring-reliability-and-accountability-from-models-to-agents.md", + "title": "\"Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents\"", + "authors": "Carol Xuan Long", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.08964", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-09", + "updated_at": "2026-05-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-09168-civex-causal-intervention-verification-for-language-agents.md", + "title": "\"CIVeX: Causal Intervention Verification for Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CIVeX: Causal Intervention Verification for Language Agents\"", + "authors": "Fabio Rovai", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.09168", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-09", + "updated_at": "2026-05-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-09692-causal-state-binding-predicts-action-control-in-language-agents.md", + "title": "Causal state binding predicts action control in language agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Causal state binding predicts action control in language agents", + "authors": "Xiao Jia", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.09692", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-10", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-10365-agent-valuebench-a-comprehensive-benchmark-for-evaluating-agent-values.md", + "title": "\"Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values\"", + "authors": "Haonan Dong, Qiguan Feng, Kehan Jiang, Haoran Ye, Xin Zhang, Guojie Song", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.10365", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-11", + "updated_at": "2026-05-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-10763-matra-modeling-the-attack-surface-of-agentic-ai-systems-openclaw-case-study.md", + "title": "\"MATRA: Modeling the Attack Surface of Agentic AI Systems -- OpenClaw Case Study\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MATRA: Modeling the Attack Surface of Agentic AI Systems -- OpenClaw Case Study\"", + "authors": "Tim Van hamme, Thomas Vissers, Javier Carnerero-Cano, Mario Fritz, Emil C. Lupu, Lieven Desmet, Dinil Mon Divakaran", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.10763", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-11", + "updated_at": "2026-05-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-10779-litmus-benchmarking-behavioral-jailbreaks-of-llm-agents-in-real-os-environments.md", + "title": "\"LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments\"", + "authors": "Chiyu Zhang, Huiqin Yang, Bendong Jiang, Xiaolei Zhang, Yiran Zhao, Ruyi Chen, Lu Zhou, Xiaogang Xu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.10779", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-11", + "updated_at": "2026-05-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-10870-remember-the-decision-not-the-description-a-rate-distortion-framework-for-agent-.md", + "title": "\"Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory\"", + "authors": "Mingxi Zou, Zhihan Guo, Langzhang Liang, Zhuo Wang, Qifan Wang, Qingsong Wen, Irwin King, Lizhen Qu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.10870", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-11", + "updated_at": "2026-05-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-11039-the-granularity-mismatch-in-agent-security-argument-level-provenance-solves-enfo.md", + "title": "\"The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck\"", + "authors": "Linfeng Fan, Ziwei Li, Yuan Tian, Yichen Wang, Rongsheng Li, Xiong Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.11039", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-11", + "updated_at": "2026-05-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-11225-pivot-bridging-planning-and-execution-in-llm-agents-via-trajectory-refinement.md", + "title": "\"PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement\"", + "authors": "Tuo Zhang, Alin-Ionut Popa, Yan Xu, Rui Song, Dimitrios Dimitriadis", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.11225", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-11", + "updated_at": "2026-05-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "autonomous-agent-llm, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-11388-deep-reasoning-in-general-purpose-agents-via-structured-meta-cognition.md", + "title": "Deep Reasoning in General Purpose Agents via Structured Meta-Cognition", + "type": "paper", + "meta": { + "type": "paper", + "title": "Deep Reasoning in General Purpose Agents via Structured Meta-Cognition", + "authors": "Dean Light, Michael Theologitis, Kshitish Ghate, Shuyue Stella Li, Benjamin Newman, Chirag Shah, Aylin Caliskan, Pang Wei Koh, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.11388", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-12", + "updated_at": "2026-05-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-11534-prism-planning-and-reasoning-with-intent-in-simulated-embodied-environments.md", + "title": "\"PRISM: : Planning and Reasoning with Intent in Simulated Embodied Environments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PRISM: : Planning and Reasoning with Intent in Simulated Embodied Environments\"", + "authors": "Yunn Kang Lim, Pengzhan Sun, Ziyi Bai, Xun Xu, Angela Yao, Xulei Yang, Shijie Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.11534", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-12", + "updated_at": "2026-05-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-11633-can-llm-agents-respond-to-disasters-benchmarking-heterogeneous-geospatial-reason.md", + "title": "Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations", + "authors": "Junjue Wang, Weihao Xuan, Heli Qi, Pengyu Dai, Kunyi Liu, Hongruixuan Chen, Zhuo Zheng, Junshi Xia, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.11633", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-12", + "updated_at": "2026-05-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-11882-on-policy-self-evolution-via-failure-trajectories-for-agentic-safety-alignment.md", + "title": "On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment", + "type": "paper", + "meta": { + "type": "paper", + "title": "On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment", + "authors": "Bo Yin, Qi Li, Xinchao Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.11882", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-12", + "updated_at": "2026-05-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-11928-when-simulation-lies-a-sim-to-real-benchmark-and-domain-randomized-rl-recipe-for.md", + "title": "\"When Simulation Lies: A Sim-to-Real Benchmark and Domain-Randomized RL Recipe for Tool-Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When Simulation Lies: A Sim-to-Real Benchmark and Domain-Randomized RL Recipe for Tool-Use Agents\"", + "authors": "Xiaolin Zhou, Aojie Yuan, Zheng Luo, Zipeng Ling, Xixiao Pan, Yicheng Gao, Haiyue Zhang, Jiate Li, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.11928", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-12", + "updated_at": "2026-05-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling, language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-11946-counterfactual-trace-auditing-of-llm-agent-skills.md", + "title": "Counterfactual Trace Auditing of LLM Agent Skills", + "type": "paper", + "meta": { + "type": "paper", + "title": "Counterfactual Trace Auditing of LLM Agent Skills", + "authors": "Xiaolin Zhou, Jinbo Liu, Li Li, Ryan A. Rossi, Xiyang Hu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.11946", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-12", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-12015-skillsafetybench-evaluating-agent-safety-under-skill-facing-attack-surfaces.md", + "title": "\"SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces\"", + "authors": "Chang Jin, An Wang, Zeming Wei, Kai Wang, Biaojie Zeng, Qiaosheng Zhang, Chao Yang, Jingjing Qu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.12015", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-12", + "updated_at": "2026-05-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-12061-sage-a-self-evolving-agentic-graph-memory-engine-for-structure-aware-associative.md", + "title": "\"SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory\"", + "authors": "Juntong Wang, Haoyue Zhao, guanghui Pan, Xiyuan Wang, Yanbo Wang, Qiyan Deng, Muhan Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.12061", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-12", + "updated_at": "2026-05-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-12260-prism-pareto-efficient-retrieval-over-intent-aware-structured-memory-for-long-ho.md", + "title": "\"PRISM: Pareto-Efficient Retrieval over Intent-Aware Structured Memory for Long-Horizon Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PRISM: Pareto-Efficient Retrieval over Intent-Aware Structured Memory for Long-Horizon Agents\"", + "authors": "Jingyi Peng, Zhongwei Wan, Weiting Liu, Qiuzhuang Sun", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.12260", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-12", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-13481-personalai-2-0-enhancing-knowledge-graph-traversal-retrieval-with-planning-mecha.md", + "title": "\"PersonalAI 2.0: Enhancing knowledge graph traversal/retrieval with planning mechanism for Personalized LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PersonalAI 2.0: Enhancing knowledge graph traversal/retrieval with planning mechanism for Personalized LLM Agents\"", + "authors": "Mikhail Menschikov, Matvey Iskornev, Alexander Kharitonov, Alina Bogdanova, Mikhail Belkin, Ekaterina Lisitsyna, Artyom Sosedka, Victoria Dochkina, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.13481", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-13", + "updated_at": "2026-05-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-13542-realicu-do-llm-agents-understand-long-context-icu-data-a-benchmark-beyond-behavi.md", + "title": "\"RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation\"", + "authors": "Chengzhi Shen, Weixiang Shen, Tobias Susetzky, Chen, Chen, Jun Li, Yuyuan Liu, Xuepeng Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.13542", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-13", + "updated_at": "2026-05-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-13618-openaaas-an-open-agent-as-a-service-framework-for-distributed-materials-informat.md", + "title": "\"OpenAaaS: An Open Agent-as-a-Service Framework for Distributed Materials-Informatics Research\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OpenAaaS: An Open Agent-as-a-Service Framework for Distributed Materials-Informatics Research\"", + "authors": "Peng Kang, Bixuan Li, Xiaoya Huang, Shuo Shi, Weiqiao Zhou, Zhen Li, Yu Liu, Lei Zheng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.13618", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-13", + "updated_at": "2026-05-13", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cond-mat.mtrl-sci", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-13716-skillops-managing-llm-agent-skill-libraries-as-self-maintaining-software-ecosyst.md", + "title": "\"SkillOps: Managing LLM Agent Skill Libraries as Self-Maintaining Software Ecosystems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SkillOps: Managing LLM Agent Skill Libraries as Self-Maintaining Software Ecosystems\"", + "authors": "Hongji Pu, Xinyuan Song, Liang Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.13716", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-13", + "updated_at": "2026-05-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14126-reinforcement-learning-for-tool-calling-agents-in-fast-healthcare-interoperabili.md", + "title": "Reinforcement Learning for Tool-Calling Agents in Fast Healthcare Interoperability Resources (FHIR)", + "type": "paper", + "meta": { + "type": "paper", + "title": "Reinforcement Learning for Tool-Calling Agents in Fast Healthcare Interoperability Resources (FHIR)", + "authors": "Marius S. Knorr, Robert Müller, Jan P. Bremer, Nils Schweingruber", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14126", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-13", + "updated_at": "2026-05-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14290-web-agents-should-adopt-the-plan-then-execute-paradigm.md", + "title": "Web Agents Should Adopt the Plan-Then-Execute Paradigm", + "type": "paper", + "meta": { + "type": "paper", + "title": "Web Agents Should Adopt the Plan-Then-Execute Paradigm", + "authors": "Julien Piet, Annabella Chow, Yiwei Hou, Muxi Lyu, Sylvie Venuto, Jinhao Zhu, Raluca Ada Popa, David Wagner", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14290", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14322-are-agents-ready-to-teach-a-multi-stage-benchmark-for-real-world-teaching-workfl.md", + "title": "Are Agents Ready to Teach? A Multi-Stage Benchmark for Real-World Teaching Workflows", + "type": "paper", + "meta": { + "type": "paper", + "title": "Are Agents Ready to Teach? A Multi-Stage Benchmark for Real-World Teaching Workflows", + "authors": "Zixin Chen, Peng Liu, Rui Sheng, Haobo Li, Jianhong Tu, Xiaodong Deng, Kashun Shum, Dayiheng Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14322", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14421-memlineage-lineage-guided-enforcement-for-llm-agent-memory.md", + "title": "\"MemLineage: Lineage-Guided Enforcement for LLM Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemLineage: Lineage-Guided Enforcement for LLM Agent Memory\"", + "authors": "Ciyan Ouyang, Rui Hou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14421", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14460-exploiting-llm-agent-supply-chains-via-payload-less-skills.md", + "title": "Exploiting LLM Agent Supply Chains via Payload-less Skills", + "type": "paper", + "meta": { + "type": "paper", + "title": "Exploiting LLM Agent Supply Chains via Payload-less Skills", + "authors": "Xinyu Liu, Yukai Zhao, Xing Hu, Xin Xia", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14460", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14498-groupmembench-benchmarking-llm-agent-memory-in-multi-party-conversations.md", + "title": "\"GroupMemBench: Benchmarking LLM Agent Memory in Multi-Party Conversations\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GroupMemBench: Benchmarking LLM Agent Memory in Multi-Party Conversations\"", + "authors": "Jingbo Yang, Kwei-Herng Lai, Xiaowen Wang, Shiyu Chang, Yaar Harari, Evgeniy Gabrilovich", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14498", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14527-lang2mlip-end-to-end-language-to-machine-learning-interatomic-potential-developm.md", + "title": "\"Lang2MLIP: End-to-End Language-to-Machine Learning Interatomic Potential Development with Autonomous Agentic Workflows\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Lang2MLIP: End-to-End Language-to-Machine Learning Interatomic Potential Development with Autonomous Agentic Workflows\"", + "authors": "Wenwen Li, Yuki Orimo, Nontawat Charoenphakdee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14527", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cond-mat.mtrl-sci", + "physics.comp-ph" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14892-beyond-individual-intelligence-surveying-collaboration-failure-attribution-and-s.md", + "title": "\"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems\"", + "authors": "Shihao Qi, Jie Ma, Rui Xing, Wei Guo, Xiao Huang, Zhitao Gao, Jianhao Deng, Jun Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14892", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14906-memlens-benchmarking-multimodal-long-term-memory-in-large-vision-language-models.md", + "title": "\"MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models\"", + "authors": "Xiyu Ren, Zhaowei Wang, Yiming Du, Zhongwei Xie, Chi Liu, Xinlin Yang, Haoyue Feng, Wenjun Pan, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14906", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-14932-toward-securing-ai-agents-like-operating-systems.md", + "title": "Toward Securing AI Agents Like Operating Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Toward Securing AI Agents Like Operating Systems", + "authors": "Lukas Pirch, Micha Horlboge, Patrick Großmann, Syeda Mahnur Asif, Klim Kireev, Thorsten Holz, Konrad Rieck", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.14932", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-15040-orchard-an-open-source-agentic-modeling-framework.md", + "title": "\"Orchard: An Open-Source Agentic Modeling Framework\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Orchard: An Open-Source Agentic Modeling Framework\"", + "authors": "Baolin Peng, Wenlin Yao, Qianhui Wu, Hao Cheng, Xiao Yu, Rui Yang, Tao Ge, Alessandro Sordoni, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.15040", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-15128-memeye-a-visual-centric-evaluation-framework-for-multimodal-agent-memory.md", + "title": "\"MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory\"", + "authors": "Minghao Guo, Qingyue Jiao, Zeru Shi, Yihao Quan, Boxuan Zhang, Danrui Li, Liwei Che, Wujiang Xu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.15128", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-14", + "updated_at": "2026-05-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.CL", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-15206-agentstop-terminating-local-ai-agents-early-to-save-energy-in-consumer-devices.md", + "title": "\"AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices\"", + "authors": "Dzung Pham, Kleomenis Katevas, Ali Shahin Shamsabadi, Hamed Haddadi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.15206", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-01", + "updated_at": "2026-05-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.DC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-15625-colpackagent-agent-skill-guided-hard-particle-monte-carlo-workflows-for-colloida.md", + "title": "\"ColPackAgent: Agent-Skill-Guided Hard-Particle Monte Carlo Workflows for Colloidal Packing\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ColPackAgent: Agent-Skill-Guided Hard-Particle Monte Carlo Workflows for Colloidal Packing\"", + "authors": "Lijie Ding, Changwoo Do", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.15625", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-15", + "updated_at": "2026-05-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cond-mat.soft" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-15701-h-mem-a-novel-memory-mechanism-for-evolving-and-retrieving-agent-memory-via-a-hy.md", + "title": "\"H-Mem: A Novel Memory Mechanism for Evolving and Retrieving Agent Memory via a Hybrid Structure\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"H-Mem: A Novel Memory Mechanism for Evolving and Retrieving Agent Memory via a Hybrid Structure\"", + "authors": "Jiawei Yu, Yixiang Fang, Xilin Liu, Yuchi Ma", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.15701", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-15", + "updated_at": "2026-05-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-15710-smmbench-a-benchmark-for-source-distributed-multimodal-agent-memory.md", + "title": "\"SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory\"", + "authors": "Huacan Chai, Yukai Wang, Yingxuan Yang, Dan Peng, Yuanyi Song, Zhihui Fu, Weiwen Liu, Jianghao Lin, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.15710", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-15", + "updated_at": "2026-05-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-15759-dimmem-dimensional-structuring-for-efficient-long-term-agent-memory.md", + "title": "\"DimMem: Dimensional Structuring for Efficient Long-Term Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DimMem: Dimensional Structuring for Efficient Long-Term Agent Memory\"", + "authors": "Wentao Qiu, Haotian Hu, Fanyi Wang, Jinwei Kong, Yu Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.15759", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-15", + "updated_at": "2026-05-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-16233-forge-self-evolving-agent-memory-with-no-weight-updates-via-population-broadcast.md", + "title": "\"FORGE: Self-Evolving Agent Memory With No Weight Updates via Population Broadcast\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"FORGE: Self-Evolving Agent Memory With No Weight Updates via Population Broadcast\"", + "authors": "Igor Bogdanov, Chung-Horng Lung, Thomas Kunz, Jie Gao, Adrian Taylor, Marzia Zaman", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.16233", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-15", + "updated_at": "2026-05-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA", + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-16481-visual-agentic-memory-enabling-online-long-video-understanding-via-online-indexi.md", + "title": "\"Visual Agentic Memory: Enabling Online Long Video Understanding via Online Indexing, Hierarchical Memory, and Agentic Retrieval\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Visual Agentic Memory: Enabling Online Long Video Understanding via Online Indexing, Hierarchical Memory, and Agentic Retrieval\"", + "authors": "Aiden Yiliu Li, Nels Numan, Anthony Steed", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.16481", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-15", + "updated_at": "2026-05-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-16821-multi-paradigm-agent-interaction-in-practice-a-systematic-analysis-of-generator-.md", + "title": "\"Multi-Paradigm Agent Interaction in Practice:A Systematic Analysis of Generator-Evaluator, ReAct Loop,and Adversarial Evaluation in the buddyMe Framework\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Multi-Paradigm Agent Interaction in Practice:A Systematic Analysis of Generator-Evaluator, ReAct Loop,and Adversarial Evaluation in the buddyMe Framework\"", + "authors": "Xiaohua Wang, Chao Han, Kai Yu, XiaoLiang Xu, Liang Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.16821", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-16", + "updated_at": "2026-05-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-17075-a-red-teaming-framework-for-evaluating-robustness-of-ai-enabled-security-orchest.md", + "title": "A Red Teaming Framework for Evaluating Robustness of AI-enabled Security Orchestration, Automation, and Response Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Red Teaming Framework for Evaluating Robustness of AI-enabled Security Orchestration, Automation, and Response Systems", + "authors": "Ayan Javeed Shaikh, Nathaniel D. Bastian, Ankit Shah", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.17075", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-16", + "updated_at": "2026-05-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-17348-taming-zombie-agents-a-markov-state-aware-framework-for-resilient-multi-agent-ev.md", + "title": "\"Taming \\\"Zombie'' Agents: A Markov State-Aware Framework for Resilient Multi-Agent Evolution\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Taming \\\"Zombie'' Agents: A Markov State-Aware Framework for Resilient Multi-Agent Evolution\"", + "authors": "Taolin Zhang, Pukun Zhao, Qizhou Chen, Jiuheng Wan, Chen Chen, Xiaofeng He, Chengyu Wang, Richang Hong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.17348", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-17", + "updated_at": "2026-05-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "memory", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-17453-trust-no-tool-evaluating-and-defending-llm-agents-under-untrusted-tool-feedback.md", + "title": "\"Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback\"", + "authors": "Lecheng Yan, Ruizhe Li, Xicheng Han, Wenxi Li, Binwu Wang, Longyue Wang, Chenyang Lyu, Guanhua Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.17453", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-17", + "updated_at": "2026-05-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-17625-episodic-semantic-memory-architecture-for-long-horizon-scientific-agents.md", + "title": "Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents", + "authors": "Nikola Milosevic", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.17625", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-17", + "updated_at": "2026-05-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-18284-commitdistill-a-lightweight-knowledge-centric-memory-layer-for-software-reposito.md", + "title": "\"CommitDistill: A Lightweight Knowledge-Centric Memory Layer for Software Repositories\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CommitDistill: A Lightweight Knowledge-Centric Memory Layer for Software Repositories\"", + "authors": "Divya Chukkapalli, Thejesh Avula, Aditya Aggarwal, Harsimran Singh, Amith Tallanki", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.18284", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-18", + "updated_at": "2026-05-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-18502-the-distance-based-formation-controller-design-for-multi-agent-systems-in-port-h.md", + "title": "The distance-based formation controller design for multi-agent systems in port-Hamiltonian form", + "type": "paper", + "meta": { + "type": "paper", + "title": "The distance-based formation controller design for multi-agent systems in port-Hamiltonian form", + "authors": "Jingyi Zhao, Yongxin Wu, Héctor García de Marina, Yuhu Wu, Yann Le Gorrec", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.18502", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-18", + "updated_at": "2026-05-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "math.OC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-18652-mementogui-learning-agentic-multimodal-memory-control-for-long-horizon-gui-agent.md", + "title": "\"MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents\"", + "authors": "Ziyun Zeng, Hang Hua, Bocheng Zou, Mu Cai, Rogerio Feris, Jiebo Luo", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.18652", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-18", + "updated_at": "2026-05-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-18672-position-a-three-layer-probabilistic-assume-guarantee-architecture-is-structural.md", + "title": "\"Position: A Three-Layer Probabilistic Assume-Guarantee Architecture Is Structurally Required for Safe LLM Agent Deployment\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Position: A Three-Layer Probabilistic Assume-Guarantee Architecture Is Structurally Required for Safe LLM Agent Deployment\"", + "authors": "S. Bensalem, Y. Dong, M. Franzle, X. Huang, J. Kroger, D. Nickovic, A. Nouri, R. Roy, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.18672", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-18", + "updated_at": "2026-05-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-18930-oep-poisoning-self-evolving-llm-agents-via-locally-correct-but-non-transferable-.md", + "title": "\"OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences\"", + "authors": "Kaixiang Wang, Jiong Lou, Zhaojiacheng Zhou, Jie Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.18930", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-18", + "updated_at": "2026-05-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-19604-formal-skill-programmable-runtime-skills-for-efficient-and-accurate-llm-agents.md", + "title": "\"Formal Skill: Programmable Runtime Skills for Efficient and Accurate LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Formal Skill: Programmable Runtime Skills for Efficient and Accurate LLM Agents\"", + "authors": "Xi Zhang, Meijun Gao, Yuntian Zhao, Xinyu Tan, Yilun Yao, Feiyu Wang, Yanshu Wang, Dingsiyi, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.19604", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-19", + "updated_at": "2026-05-19", + "status": "queued", + "relevance": "high", + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-19952-rethinking-how-to-remember-beyond-atomic-facts-in-lifelong-llm-agent-memory.md", + "title": "\"Rethinking How to Remember: Beyond Atomic Facts in Lifelong LLM Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Rethinking How to Remember: Beyond Atomic Facts in Lifelong LLM Agent Memory\"", + "authors": "Jingwei Sun, Jianing Zhu, Jiangchao Yao, Tongliang Liu, Bo Han", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.19952", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-19", + "updated_at": "2026-05-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-20306-wildroadbench-a-wild-aerial-road-damage-grounding-benchmark-for-vision-language-.md", + "title": "\"WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents\"", + "authors": "Bingnan Liu, Chenhang Cui, Rui Huang, Jiani Luo, Zhirong Shen, Tinghao Wang, Xiande Huang, Lingbei Meng, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.20306", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-19", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-20315-mix-quant-quantized-prefilling-precise-decoding-for-agentic-llms.md", + "title": "\"Mix-Quant: Quantized Prefilling, Precise Decoding for Agentic LLMs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Mix-Quant: Quantized Prefilling, Precise Decoding for Agentic LLMs\"", + "authors": "Haiquan Lu, Zigeng Chen, Gongfan Fang, Xinyin Ma, Xinchao Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.20315", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-19", + "updated_at": "2026-05-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-20616-auto-dreamer-learning-offline-memory-consolidation-for-language-agents.md", + "title": "\"Auto-Dreamer: Learning Offline Memory Consolidation for Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Auto-Dreamer: Learning Offline Memory Consolidation for Language Agents\"", + "authors": "Chongrui Ye, Yuxiang Liu, Yu Wang, Haofei Yu, Yining Zhao, Ge Liu, Julian McAuley, Jiaxuan You", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.20616", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-20", + "updated_at": "2026-05-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory, language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-20833-memgym-a-long-horizon-memory-environment-for-llm-agents.md", + "title": "\"MemGym: a Long-Horizon Memory Environment for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemGym: a Long-Horizon Memory Environment for LLM Agents\"", + "authors": "Wujiang Xu, Yu Wang, Kai Mei, Kaiqu Liang, Zhenting Wang, Mingyu Jin, Han Zhang, Shi-Xiong Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.20833", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-20", + "updated_at": "2026-05-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "26", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-20874-governance-by-construction-for-generalist-agents.md", + "title": "Governance by Construction for Generalist Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Governance by Construction for Generalist Agents", + "authors": "Segev Shlomov, Iftach Shoham, Alon Oved, Ido Levy, Sami Marreed, Harold Ship, Offer Akrabi, Sergey Zeltyn, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.20874", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-20", + "updated_at": "2026-05-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-21240-apex-autonomous-policy-exploration-for-self-evolving-llm-agents.md", + "title": "\"APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents\"", + "authors": "Yibo Li, Jiashuo Yang, Zhi Zheng, Zhiyuan Hu, Yuan Sui, Shizun Wang, Yufei He, Bryan Hooi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.21240", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-20", + "updated_at": "2026-05-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-21740-smdd-bench-can-llms-solve-real-world-small-molecule-drug-design-tasks.md", + "title": "\"SMDD-Bench: Can LLMs Solve Real-World Small Molecule Drug Design Tasks?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SMDD-Bench: Can LLMs Solve Real-World Small Molecule Drug Design Tasks?\"", + "authors": "Kevin Han, Renfei Zhang, Kathy Wei, Hamed Mahdavi, Niloofar Mireshghallah, Amir Barati Farimani", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.21740", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-20", + "updated_at": "2026-05-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-22154-idlespec-exploiting-idle-time-via-speculative-planning-for-llm-agents.md", + "title": "\"IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents\"", + "authors": "Daewon Choi, Kyunghyun Park, Woomin Song, Saket Dingliwal, Sai Muralidhar Jayanthi, Jinwoo Shin, Aram Galstyan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.22154", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-21", + "updated_at": "2026-05-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-22321-benchmarking-autonomous-agents-against-temporal-spatial-and-semantic-evasions.md", + "title": "Benchmarking Autonomous Agents against Temporal, Spatial, and Semantic Evasions", + "type": "paper", + "meta": { + "type": "paper", + "title": "Benchmarking Autonomous Agents against Temporal, Spatial, and Semantic Evasions", + "authors": "Jianan Ma, Xiaohu Du, Ruixiao Lin, Yaoxiang Bian, Jialuo Chen, Jingyi Wang, Xiaofang Yang, Shiwen Cui, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.22321", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-21", + "updated_at": "2026-05-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-22643-boiling-the-frog-a-multi-turn-benchmark-for-agentic-safety.md", + "title": "\"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety\"", + "authors": "Piercosma Bisconti, Matteo Prandi, Federico Pierucci, Federico Sartore, Enrico Panai, Laura Caroli, Yue Zhu, Adam Leon Smith, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.22643", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-21", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-23067-what-training-data-teaches-rl-memory-agents-an-empirical-study-of-curriculum-eff.md", + "title": "\"What Training Data Teaches RL Memory Agents: An Empirical Study of Curriculum Effects in Memory-Augmented QA\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"What Training Data Teaches RL Memory Agents: An Empirical Study of Curriculum Effects in Memory-Augmented QA\"", + "authors": "Xinjie He, Zhiyuan Lin, Su Liu, Jialun Wu, Qiyang Xie, Weikai Zhou, Shuai Xiao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.23067", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-21", + "updated_at": "2026-05-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-23574-push-your-agent-measuring-and-enforcing-quantitative-goal-persistence-in-long-ho.md", + "title": "\"Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents\"", + "authors": "Yuandao Cai, Yuzhang Zhu, Liyou Gao, Wensheng Tang, Shengchao Qin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.23574", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-22", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-23636-rf-instrument-agent-rfia-empowering-rf-instruments-with-natural-language-underst.md", + "title": "\"RF Instrument Agent (RFIA): Empowering RF Instruments with Natural Language Understanding, Scheduling and Execution of Complex Tasks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RF Instrument Agent (RFIA): Empowering RF Instruments with Natural Language Understanding, Scheduling and Execution of Complex Tasks\"", + "authors": "Chunhui Li, Wei Fan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.23636", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-22", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-23723-memaudit-post-hoc-auditing-of-poisoned-agent-memory-via-causal-attribution-and-s.md", + "title": "\"MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection\"", + "authors": "Zhewen Tan, Yilun Yao, Huiyan Jin, Wenhan Yu, Guoan Wang, Mengyuan Fan, liang lu, Feng Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.23723", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-22", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-23899-from-raw-experience-to-skill-consumption-a-systematic-study-of-model-generated-a.md", + "title": "\"From Raw Experience to Skill Consumption: A Systematic Study of Model-Generated Agent Skills\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Raw Experience to Skill Consumption: A Systematic Study of Model-Generated Agent Skills\"", + "authors": "Zisu Huang, Jingwen Xu, Yifan Yang, Ziyang Gong, Qihao Yang, Muzhao Tian, Xiaohua Wang, Changze Lv, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.23899", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-22", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-23986-memforest-an-efficient-agent-memory-system-with-hierarchical-temporal-indexing.md", + "title": "\"MemForest: An Efficient Agent Memory System with Hierarchical Temporal Indexing\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemForest: An Efficient Agent Memory System with Hierarchical Temporal Indexing\"", + "authors": "Han Chen, Zining Zhang, Wenqi Pei, Bingsheng He, Ming Wu, Jason Zeng, Michael Heinrich, Wei Wu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.23986", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-16", + "updated_at": "2026-05-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DB", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-24069-when-the-manual-lies-a-realistic-benchmark-to-evaluate-mcp-poisoning-attacks-for.md", + "title": "\"When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents\"", + "authors": "Shi Liu, Xuehai Tang, Xikang Yang, Liang Lin, Biyu Zhou, Wenjie Xiao, Wantao Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.24069", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-22", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-24216-agent-tom-learning-to-monitor-autonomous-llm-agents-via-theory-of-mind-reasoning.md", + "title": "\"Agent-ToM: Learning to Monitor Autonomous LLM Agents via Theory-of-Mind Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agent-ToM: Learning to Monitor Autonomous LLM Agents via Theory-of-Mind Reasoning\"", + "authors": "Nesreen K. Ahmed, Nima Nafisi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.24216", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-22", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-24219-beyond-final-answers-auditing-trajectory-level-hallucinations-in-multi-agent-ind.md", + "title": "\"Beyond Final Answers: Auditing Trajectory-Level Hallucinations in Multi-Agent Industrial Workflows\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Final Answers: Auditing Trajectory-Level Hallucinations in Multi-Agent Industrial Workflows\"", + "authors": "Harshada Badave, Santosh Borse, Andrea Gomez, Harshitha Narahari, Sara Carter, Vishwa Bhatt, Aishani Rachakonda, Shuxin Lin, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.24219", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-22", + "updated_at": "2026-05-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-24220-polar-agentic-rl-on-any-harness-at-scale.md", + "title": "\"Polar: Agentic RL on Any Harness at Scale\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Polar: Agentic RL on Any Harness at Scale\"", + "authors": "Binfeng Xu, Hao Zhang, Shaokun Zhang, Songyang Han, Mingjie Liu, Jian Hu, Shizhe Diao, Zhenghui Jin, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.24220", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-22", + "updated_at": "2026-05-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-24309-reframing-llm-agent-security-as-an-agent-human-interaction-problem.md", + "title": "Reframing LLM Agent Security as an Agent-Human Interaction Problem", + "type": "paper", + "meta": { + "type": "paper", + "title": "Reframing LLM Agent Security as an Agent-Human Interaction Problem", + "authors": "Peiran Wang, Ying Li, Yuan Tian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.24309", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-23", + "updated_at": "2026-05-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-24659-iterinject-indirect-prompt-injection-against-llm-agents-via-feedback-guided-iter.md", + "title": "\"IterInject: Indirect Prompt Injection Against LLM Agents via Feedback-Guided Iterative Optimization\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"IterInject: Indirect Prompt Injection Against LLM Agents via Feedback-Guided Iterative Optimization\"", + "authors": "Zixuan Chen, Jiaxiang Chen, Li Luo, Ke Xu, Xiaoxiang Huang, Tanfeng Sun, Xinghao Jiang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.24659", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-23", + "updated_at": "2026-05-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-24812-core-code-collaborative-reinforcement-learning-for-code-generation.md", + "title": "\"CoRe-Code: Collaborative Reinforcement Learning for Code Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CoRe-Code: Collaborative Reinforcement Learning for Code Generation\"", + "authors": "Zhihao Dou, Qinjian Zhao, Zhongwei Wan, Xiaoyu Xia, Sumon Biswas", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.24812", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-24", + "updated_at": "2026-05-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-25141-llm-agent-based-renewable-energy-forecasting-using-edge-and-iot-data-a-review-of.md", + "title": "LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support", + "type": "paper", + "meta": { + "type": "paper", + "title": "LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support", + "authors": "Pavan Manjunath, Thomas Pruefer", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.25141", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-24", + "updated_at": "2026-05-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-25200-grouptravelbench-benchmarking-llm-agents-on-multi-person-travel-planning.md", + "title": "\"GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning\"", + "authors": "Xiang Cheng, Yulan Hu, Lulu Zheng, Zheng Pan, Xin Li, Yong Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.25200", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-24", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-25310-tool-call-dependency-structure-is-linearly-decodable-in-llm-agent-residual-strea.md", + "title": "Tool-Call Dependency Structure is Linearly Decodable in LLM Agent Residual Streams", + "type": "paper", + "meta": { + "type": "paper", + "title": "Tool-Call Dependency Structure is Linearly Decodable in LLM Agent Residual Streams", + "authors": "Tianda Sun, Dimitar Kazakov", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.25310", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-25", + "updated_at": "2026-05-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-25393-decision-making-with-lightweight-confidence-aware-language-model-for-autonomous-.md", + "title": "Decision-Making with Lightweight Confidence-Aware Language Model for Autonomous Driving", + "type": "paper", + "meta": { + "type": "paper", + "title": "Decision-Making with Lightweight Confidence-Aware Language Model for Autonomous Driving", + "authors": "Ruoyu Yao, Ruiguo Zhong, Pei Liu, Mingxing Peng, Rui Yang, Jun Ma", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.25393", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-25", + "updated_at": "2026-05-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-25435-security-of-openclaw-agents-fundamentals-attacks-and-countermeasures.md", + "title": "\"Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures\"", + "authors": "Yuntao Wang, Jianle Ba, Han Liu, Yanghe Pan, Jintao Wei, Zhou Su, Tom H. Luan, Linkang Du", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.25435", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-25", + "updated_at": "2026-05-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-25920-can-llms-time-travel-enhancing-temporal-consistency-in-legal-agentic-search-thro.md", + "title": "Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning", + "authors": "Wei Fan, Yining Zhou, Mufan Zhang, Yanbing Weng, Yiran HU, Tianshi Zheng, Baixuan Xu, Chunyang Li, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.25920", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-25", + "updated_at": "2026-05-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-26165-tool-schema-compression-enables-agentic-rag-under-constrained-context-budgets.md", + "title": "Tool-Schema Compression Enables Agentic RAG Under Constrained Context Budgets", + "type": "paper", + "meta": { + "type": "paper", + "title": "Tool-Schema Compression Enables Agentic RAG Under Constrained Context Budgets", + "authors": "Furkan Sakizli", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.26165", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-24", + "updated_at": "2026-05-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-26252-is-agent-memory-a-database-rethinking-data-foundations-for-long-term-ai-agent-me.md", + "title": "Is Agent Memory a Database? Rethinking Data Foundations for Long-Term AI Agent Memory", + "type": "paper", + "meta": { + "type": "paper", + "title": "Is Agent Memory a Database? Rethinking Data Foundations for Long-Term AI Agent Memory", + "authors": "Abdelghny Orogat, Essam Mansour", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.26252", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-25", + "updated_at": "2026-05-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.DB" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-26305-experiments-in-agentic-ai-for-science.md", + "title": "Experiments in Agentic AI for Science", + "type": "paper", + "meta": { + "type": "paper", + "title": "Experiments in Agentic AI for Science", + "authors": "Judy Fox, Geoffrey Fox", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.26305", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-25", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "eess.SY", + "hep-ph" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-26497-aligning-provenance-with-authorization-a-dual-graph-defense-for-llm-agents.md", + "title": "\"Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents\"", + "authors": "Peiran Wang, Ying Li, Yuan Tian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.26497", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-05-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-26720-towards-feedback-to-plan-decisions-for-self-evolving-llm-agents-in-cuda-kernel-g.md", + "title": "Towards Feedback-to-Plan Decisions for Self-Evolving LLM Agents in CUDA Kernel Generation", + "type": "paper", + "meta": { + "type": "paper", + "title": "Towards Feedback-to-Plan Decisions for Self-Evolving LLM Agents in CUDA Kernel Generation", + "authors": "Yee Hin Chong, Jiaming Wu, Youhui Zhang, Peng Qu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.26720", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-05-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-26926-from-norms-to-indicators-n2i-rag-an-agentic-retrieval-augmented-generation-frame.md", + "title": "\"From Norms to Indicators (N2I-RAG): An Agentic Retrieval-Augmented Generation Framework for Legal Indicator Computation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Norms to Indicators (N2I-RAG): An Agentic Retrieval-Augmented Generation Framework for Legal Indicator Computation\"", + "authors": "Youssef Al Mouatamid, Marie Bonnin, Jihad Zahir", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.26926", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-05-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-27123-rethinking-agentic-rag-toward-llm-driven-logical-retrieval-beyond-embeddings.md", + "title": "\"Rethinking Agentic RAG: Toward LLM-Driven Logical Retrieval Beyond Embeddings\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Rethinking Agentic RAG: Toward LLM-Driven Logical Retrieval Beyond Embeddings\"", + "authors": "Yuqi Zeng, Qixiang Deng, Yulei Wan, Ruiquan Jiang, Xiaoqing Zheng, Xuanjing Huang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.27123", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-05-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-27134-scaling-benchmarking-and-reasoning-of-vision-language-agents-for-mobile-gui-navi.md", + "title": "Scaling, Benchmarking, and Reasoning of Vision-Language Agents for Mobile GUI Navigation", + "type": "paper", + "meta": { + "type": "paper", + "title": "Scaling, Benchmarking, and Reasoning of Vision-Language Agents for Mobile GUI Navigation", + "authors": "Heng Qu, Yike Liu, Renren Jin, Wenzong Zhang, Pengzhi Gao, Wei Liu, Jian Luan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.27134", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-05-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-27240-enpmr-bench-benchmarking-proactive-memory-retrieval-for-emotional-support-agents.md", + "title": "\"ENPMR-Bench: Benchmarking Proactive Memory Retrieval for Emotional Support Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ENPMR-Bench: Benchmarking Proactive Memory Retrieval for Emotional Support Agents\"", + "authors": "Xing Fu, Yulin Hu, Mengtong Ji, Haozhen Li, Yixin Sun, Weixiang Zhao, Yanyan Zhao, Bing Qin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.27240", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-05-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-27333-finharness-an-inline-lifecycle-safety-harness-for-finance-llm-agents.md", + "title": "\"FinHarness: An Inline Lifecycle Safety Harness for Finance LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"FinHarness: An Inline Lifecycle Safety Harness for Finance LLM Agents\"", + "authors": "Haoxuan Jia, Yang Liu, Bin Chong, Yingguang Yang, Yancheng Chen, Jiayu Liang, Qian Li, Hanning Lu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.27333", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-05-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-27366-muse-autoskill-self-evolving-agents-via-skill-creation-memory-management-and-eva.md", + "title": "\"MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation\"", + "authors": "Huawei Lin, Peng Li, Jie Song, Fuxin Jiang, Tieying Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.27366", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-27690-traces-proactive-safety-auditing-for-multi-turn-llm-agents-via-trajectory-state-.md", + "title": "\"TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling\"", + "authors": "Jiaqian Li, Yanshu Li, Boxuan Zhang, Ruixiang Tang, Kuan-Hao Huang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.27690", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-05-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-27762-peam-parametric-embodied-agent-memory-through-contrastive-internalization-of-exp.md", + "title": "\"PEAM: Parametric Embodied Agent Memory through Contrastive Internalization of Experience in Minecraft\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PEAM: Parametric Embodied Agent Memory through Contrastive Internalization of Experience in Minecraft\"", + "authors": "Yuchen Guo, Junli Gong, Weicheng Wang, Hongmin Cai, Yiu-ming Cheung, Weifeng Su", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.27762", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-26", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-27825-mrmmia-membership-inference-attacks-on-memory-in-chat-agents.md", + "title": "\"MRMMIA: Membership Inference Attacks on Memory in Chat Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MRMMIA: Membership Inference Attacks on Memory in Chat Agents\"", + "authors": "Kai Chen, Yan Pang, Tianhao Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.27825", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-27", + "updated_at": "2026-05-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-27935-do-agents-think-deeper-a-mechanistic-investigation-of-layer-wise-dynamics-in-seq.md", + "title": "Do Agents Think Deeper? A Mechanistic Investigation of Layer-Wise Dynamics in Sequential Planning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Do Agents Think Deeper? A Mechanistic Investigation of Layer-Wise Dynamics in Sequential Planning", + "authors": "Zhenyu Cui, Xiangzhong Luo", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.27935", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-27", + "updated_at": "2026-05-27", + "status": "queued", + "relevance": "high", + "topics": [ + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-28046-memcog-from-memory-as-tool-to-memory-as-cognition-in-conversational-agents.md", + "title": "\"MemCog: From Memory-as-Tool to Memory-as-Cognition in Conversational Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemCog: From Memory-as-Tool to Memory-as-Cognition in Conversational Agents\"", + "authors": "Zihan Li, Xingyu Fan, Feifei Li, Wenhui Que", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.28046", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-27", + "updated_at": "2026-05-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-28120-legalgraphrag-multi-agent-graph-retrieval-augmented-generation-for-reliable-lega.md", + "title": "\"LegalGraphRAG: Multi-Agent Graph Retrieval-Augmented Generation for Reliable Legal Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LegalGraphRAG: Multi-Agent Graph Retrieval-Augmented Generation for Reliable Legal Reasoning\"", + "authors": "Zerui Chen, Qinggang Zhang, Zhishang Xiang, Zhimin Wei, Linfeng Gao, Xiao Huang, Zhihong Zhang, Jinsong Su", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.28120", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-27", + "updated_at": "2026-05-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-28175-mixture-of-experts-knowledge-graph-retrieval-augmented-generation-for-multi-agen.md", + "title": "Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation", + "type": "paper", + "meta": { + "type": "paper", + "title": "Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation", + "authors": "Shijie Wang, Chengyi Liu, Yujuan Ding, Shanru Lin, See-Kiong Ng, Xu Xin, Wenqi Fan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.28175", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-27", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-28424-skill0-5-joint-skill-internalization-and-utilization-for-out-of-distribution-gen.md", + "title": "\"Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning\"", + "authors": "Jiapeng Zhu, Jianxiang Yu, Yibo Zhao, Chengcheng Han, Qi Gu, Xunliang Cai, Xiang Li, Weining Qian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.28424", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-27", + "updated_at": "2026-05-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-28607-adaptive-multimodal-agents-based-framework-for-automatic-workflow-execution.md", + "title": "Adaptive Multimodal Agents-Based Framework for Automatic Workflow Execution", + "type": "paper", + "meta": { + "type": "paper", + "title": "Adaptive Multimodal Agents-Based Framework for Automatic Workflow Execution", + "authors": "Susanna Cifani, Mario Luca Bernardi, Marta Cimitile", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.28607", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-27", + "updated_at": "2026-05-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "multi-agent", + "planning", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-28617-lacuna-safe-agents-as-recursive-program-holes.md", + "title": "\"LACUNA: Safe Agents as Recursive Program Holes\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LACUNA: Safe Agents as Recursive Program Holes\"", + "authors": "Yaoyu Zhao, Yichen Xu, Oliver Bračevac, Cao Nguyen Pham, Frank Zhengqing Wu, Martin Odersky", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.28617", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-27", + "updated_at": "2026-05-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.PL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-28787-do-agents-need-semantic-metadata-a-comparative-study-in-agentic-data-retrieval.md", + "title": "Do Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval", + "type": "paper", + "meta": { + "type": "paper", + "title": "Do Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval", + "authors": "Shiyu Chen, Tarfah Alrashed, Alon Halevy, Natasha Noy", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.28787", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-27", + "updated_at": "2026-05-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-28835-genesisfunc-multi-agent-data-generation-for-accurate-and-generalizable-function-.md", + "title": "\"GenesisFunc: Multi-Agent Data Generation for Accurate and Generalizable Function-Calling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GenesisFunc: Multi-Agent Data Generation for Accurate and Generalizable Function-Calling\"", + "authors": "Hao-Xiang Xu, Chong Deng, Jiaqing Liu, Wen Wang, Qian Chen, Lujia Bao, Xiangang Li, Zhen-Hua Ling", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.28835", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-10", + "updated_at": "2026-04-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-28850-representation-signatures-and-risk-feedback-alignment-in-llm-trading-agents.md", + "title": "Representation Signatures and Risk-Feedback Alignment in LLM Trading Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Representation Signatures and Risk-Feedback Alignment in LLM Trading Agents", + "authors": "Weicheng Xue", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.28850", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-16", + "updated_at": "2026-05-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "memory", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "q-fin.CP" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-29341-worldmemarena-evaluating-multimodal-agent-memory-through-action-world-interactio.md", + "title": "\"WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction\"", + "authors": "Chengzhi Liu, Yuzhe Yang, Sophia Xiao Pu, Yepeng Liu, Lin Long, Yichen Guo, Nuo Chen, Zhaotian Weng, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.29341", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-29630-entity-collision-a-stratified-protocol-for-attributing-retrieval-lift-in-agent-m.md", + "title": "\"Entity-Collision: A Stratified Protocol for Attributing Retrieval Lift in Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Entity-Collision: A Stratified Protocol for Attributing Retrieval Lift in Agent Memory\"", + "authors": "Youwang Deng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.29630", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-29640-vikingmem-a-memory-base-management-system-for-stateful-llm-based-applications.md", + "title": "\"VikingMem: A Memory Base Management System for Stateful LLM-based Applications\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VikingMem: A Memory Base Management System for Stateful LLM-based Applications\"", + "authors": "Jiajie Fu, Junwen Chen, Mengzhao Wang, Aoxiang He, Maojia Sheng, Xiangyu Ke, Yifan Zhu, Yunjun Gao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.29640", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-29653-ptcg-bench-can-llm-agents-master-pok-mon-trading-card-game.md", + "title": "\"PTCG-Bench: Can LLM Agents Master Pokémon Trading Card Game?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PTCG-Bench: Can LLM Agents Master Pokémon Trading Card Game?\"", + "authors": "Dongdong Hua, Yifei Sun, Renhong Huang, Feng Gao, Chunping Wang, Yang Yang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.29653", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation, autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-29676-notation-matters-a-benchmark-study-of-token-optimized-formats-in-agentic-ai-syst.md", + "title": "\"Notation Matters: A Benchmark Study of Token-Optimized Formats in Agentic AI Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Notation Matters: A Benchmark Study of Token-Optimized Formats in Agentic AI Systems\"", + "authors": "Lorenz Kutschka, Bernhard Geiger", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.29676", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-29790-evolve-as-a-team-collaborative-self-evolution-for-llm-based-multi-agent-systems.md", + "title": "\"Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems\"", + "authors": "Zhezheng Hao, Tianfu Wang, Huanshuo Dong, Ziyan Liu, Hong Wang, Xiankun Lin, Qiang Lin, Can Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.29790", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-29801-agentdog-1-5-a-lightweight-and-scalable-alignment-framework-for-ai-agent-safety-.md", + "title": "\"AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security\"", + "authors": "Dongrui Liu, Yu Li, Zhonghao Yang, Peng Wang, Guanxu Chen, Yuejin Xie, Qinghua Mao, Wanying Qu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.29801", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.CR", + "cs.CV", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-29861-towards-verifiable-multimodal-deep-research-a-multi-agent-harness-for-interleave.md", + "title": "\"Towards Verifiable Multimodal Deep Research: A Multi-Agent Harness for Interleaved Report Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Towards Verifiable Multimodal Deep Research: A Multi-Agent Harness for Interleaved Report Generation\"", + "authors": "Chenghao Zhang, Guanting Dong, Yufan Liu, Tong Zhao, Xiaoxi Li, Zhicheng Dou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.29861", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-29960-hijacking-agent-memory-stealthy-trojan-attacks-through-conversational-interactio.md", + "title": "\"Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction\"", + "authors": "Hongtao Wang, Se Yang, Yu Chen, Puzhuo Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.29960", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30058-heart-bench-do-llm-agents-exhibit-human-like-psychology.md", + "title": "\"HEART-Bench: Do LLM Agents Exhibit Human-like Psychology?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HEART-Bench: Do LLM Agents Exhibit Human-like Psychology?\"", + "authors": "Weihan Peng, Chenxu Zhang, Qianao Wang, Yuling Shi, Heng Lian, Qihong Mao, Jiahao Pang, Chunliang Feng, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30058", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30090-directorbench-diagnosing-long-form-video-generation-with-personalized-multi-agen.md", + "title": "\"DirectorBench: Diagnosing Long-Form Video Generation with Personalized Multi-Agent Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DirectorBench: Diagnosing Long-Form Video Generation with Personalized Multi-Agent Evaluation\"", + "authors": "Jiamin Chen, Qianben Chen, Jiawen Zhang, Yidi Wu, Yuchen Li, Xiaokun Zhang, Wangchunshu Zhou, Chen Ma", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30090", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30407-exploring-autonomous-agentic-data-engineering-for-model-specialization.md", + "title": "Exploring Autonomous Agentic Data Engineering for Model Specialization", + "type": "paper", + "meta": { + "type": "paper", + "title": "Exploring Autonomous Agentic Data Engineering for Model Specialization", + "authors": "Yujie Luo, Xiangyuan Ru, Jingsheng Zheng, Jingjing Wang, Yuqi Zhu, Jintian Zhang, Runnan Fang, Kewei Xu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30407", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.IR", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30604-an-organization-scoped-llm-agent-runtime-architecture-for-regulated-cybersecurit.md", + "title": "An Organization-Scoped LLM Agent Runtime Architecture for Regulated Cybersecurity Operations", + "type": "paper", + "meta": { + "type": "paper", + "title": "An Organization-Scoped LLM Agent Runtime Architecture for Regulated Cybersecurity Operations", + "authors": "George Fatouros, Georgios Makridis, George Kousiouris, John Soldatos, Dimosthenis Kyriazis", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30604", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30690-elasticmem-latent-memory-as-a-learnable-resource-for-llm-agents.md", + "title": "\"ElasticMem: Latent Memory as a Learnable Resource for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ElasticMem: Latent Memory as a Learnable Resource for LLM Agents\"", + "authors": "Tao Feng, Chongrui Ye, Tianyang Luo, Jingjun Xu, Xueqiang Xu, Haozhen Zhang, Ge Liu, Jiaxuan You", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30690", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30711-sage-a-novelty-gate-for-efficient-memory-evolution-in-agentic-llms.md", + "title": "\"SAGE: A Novelty Gate for Efficient Memory Evolution in Agentic LLMs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SAGE: A Novelty Gate for Efficient Memory Evolution in Agentic LLMs\"", + "authors": "Sijia Wang, Dhanajit Brahma, Ricardo Henao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30711", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.LG", + "stat.ML" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30858-forecastcompass-guiding-agentic-forecasting-with-adaptive-factor-memory.md", + "title": "\"ForecastCompass: Guiding Agentic Forecasting with Adaptive Factor Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ForecastCompass: Guiding Agentic Forecasting with Adaptive Factor Memory\"", + "authors": "Yurui Chang, Yongkang Du, Yuanpu Cao, Jinghui Chen, Lu Lin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30858", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30883-trace-task-aware-adaptive-self-evolving-agentic-jailbreaking.md", + "title": "\"TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking\"", + "authors": "Churui Zeng, Weiwei Qi, Kedong Xiu, Tianhang Zheng, Chaochao Lu, Liang He, Zhan Qin, Kui Ren", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30883", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30907-bluefin-benchmarking-llm-agents-on-financial-spreadsheets.md", + "title": "\"BlueFin: Benchmarking LLM Agents on Financial Spreadsheets\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"BlueFin: Benchmarking LLM Agents on Financial Spreadsheets\"", + "authors": "Srivatsa Kundurthy, Clara Na, Colton Moraine, Anoushka Mohta, Case Winter, George Fang, John Ling, Emma Strubell, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30907", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-30947-extending-ai-for-research-to-the-humanities-a-multi-agent-framework-for-evidence.md", + "title": "\"Extending AI for Research to the Humanities: A Multi-Agent Framework for Evidence-Grounded Scholarship\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Extending AI for Research to the Humanities: A Multi-Agent Framework for Evidence-Grounded Scholarship\"", + "authors": "Yating Pan, Jiajun Zhang, Jun Wang, Qi Su", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.30947", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-31075-task-focused-memorization-for-multimodal-agents.md", + "title": "Task-Focused Memorization for Multimodal Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Task-Focused Memorization for Multimodal Agents", + "authors": "Tao Zou, Yichen He, Tian Qiu, Yuan Lin, Hang Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.31075", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-31268-mellum2-technical-report.md", + "title": "Mellum2 Technical Report", + "type": "paper", + "meta": { + "type": "paper", + "title": "Mellum2 Technical Report", + "authors": "Marko Kojic, Ivan Bondyrev, Aral de Moor, Joseph Shtok, Petr Borovlev, Kseniia Lysaniuk, Madeeswaran Kannan, Ivan Dolgov, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.31268", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-31278-industrializing-prediction-powered-inference-the-glide-library-for-reliable-gena.md", + "title": "\"Industrializing Prediction-Powered Inference: The GLIDE Library for Reliable GenAI and Agentic Systems Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Industrializing Prediction-Powered Inference: The GLIDE Library for Reliable GenAI and Agentic Systems Evaluation\"", + "authors": "Grégoire Martinon, Ibrahim Merad, Mohammed Raki", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.31278", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG", + "stat.ME" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-31308-tracegraph-shared-decision-landscapes-for-diagnosing-and-improving-agent-traject.md", + "title": "\"TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories\"", + "authors": "Junjie Nian, Kang Chen, Ge Zhang, Yixin Cao, Yugang Jiang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.31308", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "embodied-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2605-31377-dynatree-dynamic-agentic-retrieval-tree-for-time-sensitive-news-retrieval.md", + "title": "\"DynaTree: Dynamic Agentic Retrieval Tree for Time-Sensitive News Retrieval\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DynaTree: Dynamic Agentic Retrieval Tree for Time-Sensitive News Retrieval\"", + "authors": "Siyuan Qi, Xinyuan Wang, Yingxuan Yang, Haochuan Guo, Jianghao Lin, Weiwen Liu, Yong Yu, Weinan Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2605.31377", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00198-bagen-are-llm-agents-budget-aware.md", + "title": "\"BAGEN: Are LLM Agents Budget-Aware?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"BAGEN: Are LLM Agents Budget-Aware?\"", + "authors": "Yuxiang Lin, Zihan Wang, Mengyang Liu, Yuxuan Shan, Longju Bai, Junyao Zhang, Xing Jin, Boshan Chen, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00198", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00341-rogue-misaligned-agent-behavior-arising-from-ordinary-computer-use.md", + "title": "\"ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use\"", + "authors": "Jeremy Tien, Abishek Anand, Yu-Rou Tuan, Yuchen Shen, J. Zico Kolter, Aran Nayebi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00341", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00610-memgraphrag-memory-based-multi-agent-system-for-graph-retrieval-augmented-genera.md", + "title": "\"MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation\"", + "authors": "Chuanjie Wu, Zhishang Xiang, Yunbo Tang, Zerui Chen, Qinggang Zhang, Jinsong Su", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00610", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-30", + "updated_at": "2026-05-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00611-trace-trajectory-risk-aware-compression-for-long-horizon-agent-safety.md", + "title": "\"TRACE: Trajectory Risk-Aware Compression for Long-Horizon Agent Safety\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TRACE: Trajectory Risk-Aware Compression for Long-Horizon Agent Safety\"", + "authors": "Zhepei Hong, Lin Wang, Liting Li, Haokai Ma, Junfeng Fang, Fei Shen, Dan Zhang, Xiang Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00611", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-30", + "updated_at": "2026-05-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00619-mempro-agentic-memory-systems-as-evolvable-programs.md", + "title": "\"MemPro: Agentic Memory Systems as Evolvable Programs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemPro: Agentic Memory Systems as Evolvable Programs\"", + "authors": "Qingshan Liu, Guoqing Wang, Wen Wu, Jingqi Huang, Xinqi Tao, Dejia Song, Jie Zhou, Liang He", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00619", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-30", + "updated_at": "2026-05-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00644-foresci-evaluating-llm-agents-for-forward-looking-ai-research-judgment.md", + "title": "\"ForeSci: Evaluating LLM Agents for Forward-Looking AI Research Judgment\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ForeSci: Evaluating LLM Agents for Forward-Looking AI Research Judgment\"", + "authors": "Qiuyu Tian, Haojie Yin, Yingce Xia, Youyong Kong, Zequn Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00644", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-30", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00756-comic-collaborative-memory-and-insights-circulation-for-long-horizon-llm-agents-.md", + "title": "\"CoMIC: Collaborative Memory and Insights Circulation for Long-Horizon LLM Agents in Cloud-Edge Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CoMIC: Collaborative Memory and Insights Circulation for Long-Horizon LLM Agents in Cloud-Edge Systems\"", + "authors": "Yannan Wang, Longli Yang, Zhen Liu, Abhishek Kumar, Carsten Maple", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00756", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-30", + "updated_at": "2026-05-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00914-adversarial-feeds-steer-llm-agent-decisions-against-their-defaults.md", + "title": "Adversarial Feeds Steer LLM Agent Decisions Against Their Defaults", + "type": "paper", + "meta": { + "type": "paper", + "title": "Adversarial Feeds Steer LLM Agent Decisions Against Their Defaults", + "authors": "Rana Muhammad Usman", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00914", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-30", + "updated_at": "2026-05-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00915-autonomous-agentic-design-for-photonics.md", + "title": "Autonomous agentic design for photonics", + "type": "paper", + "meta": { + "type": "paper", + "title": "Autonomous agentic design for photonics", + "authors": "Prashanta Kharel, Amin Khavasi, Xinzhong Chen, Tyler W. Hughes", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00915", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-30", + "updated_at": "2026-05-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "physics.optics" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00922-a-machine-to-machine-knowledge-guided-llm-agent-for-generalizable-radiotherapy-t.md", + "title": "A Machine-to-Machine Knowledge-Guided LLM Agent for Generalizable Radiotherapy Treatment Planning", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Machine-to-Machine Knowledge-Guided LLM Agent for Generalizable Radiotherapy Treatment Planning", + "authors": "Md Mainul Abrar, Xun Jia, Yujie Chi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00922", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-30", + "updated_at": "2026-05-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "physics.med-ph", + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-00939-fincom-a-financial-multi-agent-demo-with-disagree-or-commit-deliberation.md", + "title": "\"FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation\"", + "authors": "Chao Peter Yang, Zixiao Tan, Kaisen Yao, Ziyu Zhou, Eleanor Jiang, Michael Wu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.00939", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-31", + "updated_at": "2026-05-31", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01041-expweaver-llm-agents-learn-from-experience-via-latent-rag.md", + "title": "\"ExpWeaver: LLM Agents Learn from Experience via Latent RAG\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ExpWeaver: LLM Agents Learn from Experience via Latent RAG\"", + "authors": "Tao Feng, Tianyang Luo, Jingjun Xu, Zhigang Hua, Yan Xie, Shuang Yang, Ge Liu, Jiaxuan You", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01041", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-31", + "updated_at": "2026-05-31", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01138-memorywire-a-vendor-neutral-wire-format-for-agent-memory-operations.md", + "title": "\"memorywire: A Vendor-Neutral Wire Format for Agent Memory Operations\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"memorywire: A Vendor-Neutral Wire Format for Agent Memory Operations\"", + "authors": "Thamilvendhan Munirathinam", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01138", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-31", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.DC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01166-braveguard-from-open-world-threats-to-safer-computer-use-agents.md", + "title": "\"BraveGuard: From Open-World Threats to Safer Computer-Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"BraveGuard: From Open-World Threats to Safer Computer-Use Agents\"", + "authors": "Yunhao Feng, Xiaohu Du, Xinhao Deng, Yifan Ding, Ming Wen, Yixu Wang, Yuxiang Xie, Baihui Zheng, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01166", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-31", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01185-skill-issues-data-centric-optimization-of-lakehouse-agents.md", + "title": "\"\\\"Skill issues'': data-centric optimization of lakehouse agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"\\\"Skill issues'': data-centric optimization of lakehouse agents\"", + "authors": "Nicole Rose Schneider, Davide Ghilardi, Giacomo Piccinini, Jacopo Tagliabue", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01185", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-31", + "updated_at": "2026-05-31", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01199-can-llm-agents-sustain-long-horizon-organizational-dynamics.md", + "title": "Can LLM Agents Sustain Long-Horizon Organizational Dynamics?", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can LLM Agents Sustain Long-Horizon Organizational Dynamics?", + "authors": "Xuancheng Zhu, Yang Yue, Shuaibing Wan, Zihan Dou, Xiaohan Zhang, Yongrui Liu, Guoshun Nan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01199", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-31", + "updated_at": "2026-05-31", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "language-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01222-rag-driven-multi-agent-llm-framework-with-task-decomposition-for-beyond-5g-auto-.md", + "title": "RAG-driven Multi-Agent LLM Framework with Task Decomposition for Beyond 5G Auto-Configuration", + "type": "paper", + "meta": { + "type": "paper", + "title": "RAG-driven Multi-Agent LLM Framework with Task Decomposition for Beyond 5G Auto-Configuration", + "authors": "İrşat Emin Sarıdaş, Onur Salan, Ali Görçin, Ibrahim Hokelek, Hakan Ali Çırpan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01222", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-31", + "updated_at": "2026-05-31", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SP" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01385-bridging-requirements-and-architecture-multi-agent-orchestration-with-external-k.md", + "title": "\"Bridging Requirements and Architecture: Multi-Agent Orchestration with External Knowledge and Hierarchical Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Bridging Requirements and Architecture: Multi-Agent Orchestration with External Knowledge and Hierarchical Memory\"", + "authors": "Ruiyin Li, Yiran Zhang, Xiyu Zhou, Yangxiao Cai, Peng Liang, Weisong Sun, Jifeng Xuan, Zhi Jin, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01385", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-31", + "updated_at": "2026-05-31", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "multi-agent", + "rag", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01416-self-healing-agentic-orchestrators-for-reliable-tool-augmented-large-language-mo.md", + "title": "Self-Healing Agentic Orchestrators for Reliable Tool-Augmented Large Language Model Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Self-Healing Agentic Orchestrators for Reliable Tool-Augmented Large Language Model Systems", + "authors": "Rahul Suresh Babu, Adarsh Agrawal", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01416", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-31", + "updated_at": "2026-05-31", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01528-joint-agent-memory-and-exploration-learning-via-novelty-signals.md", + "title": "Joint Agent Memory and Exploration Learning via Novelty Signals", + "type": "paper", + "meta": { + "type": "paper", + "title": "Joint Agent Memory and Exploration Learning via Novelty Signals", + "authors": "Shizuo Tian, Xiaohong Weng, Rui Kong, Yuxuan Chen, Guohong Liu, Yuebing Song, Jiacheng Liu, Yuchen Li, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01528", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01613-techrag-evidence-gated-multimodal-agentic-rag-for-technical-literature-reasoning.md", + "title": "\"TechRAG: Evidence-Gated Multimodal Agentic RAG for Technical Literature Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TechRAG: Evidence-Gated Multimodal Agentic RAG for Technical Literature Reasoning\"", + "authors": "Kanwar Bharat Singh", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01613", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01815-crab-bench-evaluating-llm-agents-under-complex-task-dependencies-and-human-align.md", + "title": "\"CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation\"", + "authors": "Danqing Wang, Akshay Sivaraman, Lei Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01815", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-01961-automedbench-towards-medical-autoresearch-with-agentic-ai-models.md", + "title": "\"AutoMedBench: Towards Medical AutoResearch with Agentic AI Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AutoMedBench: Towards Medical AutoResearch with Agentic AI Models\"", + "authors": "Junqi Liu, Selena Song, Yuhan Wang, Jiawei Mao, Hardy Chen, Xiaoke Huang, Tianhao Qi, Pengfei Guo, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.01961", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02109-badger-bridging-agentic-and-deterministic-evaluation-for-generative-enterprise-r.md", + "title": "\"BADGER: Bridging Agentic and Deterministic Evaluation for Generative Enterprise Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"BADGER: Bridging Agentic and Deterministic Evaluation for Generative Enterprise Reasoning\"", + "authors": "Shannon Serrao, Soumitra Chatterjee, Dorina Strori, Abhishek Sharma, Nathan Miller", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02109", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02302-seclaw-spec-driven-security-task-synthesis-for-evaluating-autonomous-agents.md", + "title": "\"SeClaw: Spec-Driven Security Task Synthesis for Evaluating Autonomous Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SeClaw: Spec-Driven Security Task Synthesis for Evaluating Autonomous Agents\"", + "authors": "Hao Cheng, Changtao Miao, Tianle Song, Yin Wu, He Liu, Erjia Xiao, Junchi Chen, Xiaoyu Shi, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02302", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-safety, autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02372-comap-co-evolving-world-models-and-agent-policies-for-llm-agents.md", + "title": "\"COMAP: Co-Evolving World Models and Agent Policies for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"COMAP: Co-Evolving World Models and Agent Policies for LLM Agents\"", + "authors": "Youwei Liu, Jian Wang, Hanlin Wang, Wenjie Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02372", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02380-spade-bench-evaluating-spontaneous-strategic-deception-in-agents-via-plan-action.md", + "title": "\"SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence\"", + "authors": "Yuyan Bu, Haowei Li, Qirui Zheng, Bowen Dong, Kaiyue Yang, Jiaming Ji, Yingshui Tan, Wenxin Li, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02380", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02388-policy-and-world-modeling-co-training-for-language-agents.md", + "title": "Policy and World Modeling Co-Training for Language Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Policy and World Modeling Co-Training for Language Agents", + "authors": "Ning Lu, Baijiong Lin, Shengcai Liu, Jiahao Wu, Haoze Lv, Yanbin Wei, Lingting Zhu, Shengju Qian, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02388", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02404-k-browsecomp-a-web-browsing-agent-benchmark-grounded-in-korean-contexts.md", + "title": "\"K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts\"", + "authors": "Nahyun Lee, Dongkeun Yoon, Guijin Son, Geewook Kim, Dayoon Ko, Jeonghun Park, Haneul Yoo, Jaewon Cho, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02404", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02461-agentcl-toward-rigorous-evaluation-of-continual-learning-in-language-agents.md", + "title": "\"AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents\"", + "authors": "Yiheng Shu, Bernal Jiménez Gutiérrez, Saisri Padmaja Jonnalagedda, Yuguang Yao, Huan Sun, Yu Su", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02461", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02497-bridging-the-last-mile-of-time-series-forecasting-with-llm-agents.md", + "title": "Bridging the Last Mile of Time Series Forecasting with LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Bridging the Last Mile of Time Series Forecasting with LLM Agents", + "authors": "Yuhua Liao, Zetian Wang, Qiangqiang Nie, Zhenhua Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02497", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02812-traj-evolve-a-self-evolving-multi-agent-system-for-patient-trajectory-modeling-i.md", + "title": "\"Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection\"", + "authors": "Sihang Zeng, Matthew Thompson, Ruth Etzioni, Meliha Yetisgen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02812", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-02965-what-benchmarks-don-t-measure-the-case-for-evaluating-abstention-competence-in-a.md", + "title": "\"What Benchmarks Don't Measure: The Case for Evaluating Abstention Competence in Autonomous Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"What Benchmarks Don't Measure: The Case for Evaluating Abstention Competence in Autonomous Agents\"", + "authors": "Victor Ojewale, Suresh Venkatasubramanian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.02965", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-03108-evotrainer-co-evolving-llm-policies-and-training-harnesses-for-autonomous-agenti.md", + "title": "\"EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning\"", + "authors": "Guhong Chen, Yingcheng Shi, Yongbin Li, Binhua Li, Xander Xu, Hu Wei, Shiwen Ni, Min Yang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.03108", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-03135-uncertainty-aware-clarification-in-llm-agents-with-information-gain.md", + "title": "Uncertainty-Aware Clarification in LLM Agents with Information Gain", + "type": "paper", + "meta": { + "type": "paper", + "title": "Uncertainty-Aware Clarification in LLM Agents with Information Gain", + "authors": "Mengyi Deng, Zhiwei Li, Xin Li, Tingyu Zhu, Ying Zhao, Zhijiang Guo, Wei Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.03135", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-03157-clinicalmc-a-benchmark-for-multi-course-clinical-decision-making-with-large-lang.md", + "title": "\"ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models\"", + "authors": "Ruihui Hou, Siyi Zhu, Ziyue Huai, Guangya Yu, Yongqi Fan, Chunming Wang, Tong Ruan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.03157", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-03197-memtrain-self-supervised-context-memory-training.md", + "title": "\"MemTrain: Self-Supervised Context Memory Training\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemTrain: Self-Supervised Context Memory Training\"", + "authors": "Ziheng Li, Xingrun Xing, Haoqing Wang, Zhi-Hong Deng, Yehui Tang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.03197", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-03329-infomem-training-long-context-memory-agents-with-answer-conditioned-information-.md", + "title": "\"InfoMem: Training Long-Context Memory Agents with Answer-Conditioned Information Gain\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"InfoMem: Training Long-Context Memory Agents with Answer-Conditioned Information Gain\"", + "authors": "Tiancheng Han, Yong Li, Wuzhou Yu, Qiaosheng Zhang, Wenqi Shao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.03329", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-03374-emem-a-hybrid-spatio-temporal-memory-system-for-embodied-agents.md", + "title": "\"eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents\"", + "authors": "A. Haroon Rasheed, Maria Kabtoul", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.03374", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-03544-sage-a-quantitative-evaluation-of-socialized-evolution-in-agent-ecosystems.md", + "title": "\"SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems\"", + "authors": "Linyue Pan, Yaoming Zhu, Lin Qiu, Xuezhi Cao, Xunliang Cai", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.03544", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-03657-diagnosing-knowledge-gaps-in-llm-tool-use-an-agentic-benchmark-for-novel-api-acq.md", + "title": "\"Diagnosing Knowledge Gaps in LLM Tool Use: An Agentic Benchmark for Novel API Acquisition\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Diagnosing Knowledge Gaps in LLM Tool Use: An Agentic Benchmark for Novel API Acquisition\"", + "authors": "Jinnuo Liu, Yue Peng, Jinhan Niu, Hongyi Wen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.03657", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-03895-agent-libos-a-runtime-substrate-for-capability-controlled-self-evolving-llm-agen.md", + "title": "\"Agent libOS: A Runtime Substrate for Capability-Controlled Self-Evolving LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agent libOS: A Runtime Substrate for Capability-Controlled Self-Evolving LLM Agents\"", + "authors": "Yingqi Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.03895", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.OS", + "cs.AI", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04051-rubas-rubric-based-reinforcement-learning-for-agent-safety.md", + "title": "\"RUBAS: Rubric-Based Reinforcement Learning for Agent Safety\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RUBAS: Rubric-Based Reinforcement Learning for Agent Safety\"", + "authors": "Xian Qi Loye, Qinglin Su, Zhexin Zhang, Shiyao Cui, Qi Zhu, Fei Mi, Hongning Wang, Minlie Huang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04051", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04120-salimory-orchestrating-cognitive-memory-for-conversational-agents.md", + "title": "\"SaliMory: Orchestrating Cognitive Memory for Conversational Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SaliMory: Orchestrating Cognitive Memory for Conversational Agents\"", + "authors": "Kai Zhang, Xinyuan Zhang, Hongda Jiang, Shiun-Zu Kuo, Hyokun Yun, Ejaz Ahmed, Shereen Oraby, Ziyun Li, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04120", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04296-the-saturation-trap-and-the-subjectivity-of-intervention-timing-why-affect-based.md", + "title": "\"The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents\"", + "authors": "Manvendra Modgil", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04296", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-02", + "updated_at": "2026-06-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04315-exploring-cross-scenario-generality-of-agentic-memory-systems-diagnostics-and-a-.md", + "title": "\"Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline\"", + "authors": "Zhikai Chen, Jialiang Gu, Junyu Yin, Xianxuan Long, Shenglai Zeng, Xiaoze Liu, Kai Guo, Keren Zhou, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04315", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04555-temporal-order-matters-for-agentic-memory-segment-trees-for-long-horizon-agents.md", + "title": "\"Temporal Order Matters for Agentic Memory: Segment Trees for Long-Horizon Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Temporal Order Matters for Agentic Memory: Segment Trees for Long-Horizon Agents\"", + "authors": "Yifan Simon Liu, Liam Gallagher, Faeze Moradi Kalarde, Jiazhou Liang, Armin Toroghi, Scott Sanner", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04555", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04599-plan-first-judge-later-run-better-a-dmaic-inspired-agentic-system-for-industrial.md", + "title": "\"Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection\"", + "authors": "Yongzi Yu, Ao Li, Le Wang, Ziyue Li, Fugee Tsung, Yuxuan Liang, Man Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04599", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "multi-agent", + "planning", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04628-rampart-registry-based-agentic-memory-with-priority-aware-runtime-transformation.md", + "title": "\"RAMPART: Registry-based Agentic Memory with Priority-Aware Runtime Transformation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RAMPART: Registry-based Agentic Memory with Priority-Aware Runtime Transformation\"", + "authors": "Nikodem Tomczak", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04628", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04780-personatree-structured-lifecycle-memory-for-person-understanding-in-llm-agents.md", + "title": "\"PersonaTree: Structured Lifecycle Memory for Person Understanding in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PersonaTree: Structured Lifecycle Memory for Person Understanding in LLM Agents\"", + "authors": "Yubo Hou, Jingwei Song, Hongbo Zhang, Zhisheng Chen, Bang Xiao, Tao Wan, Zengchang Qin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04780", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04874-agent-planning-benchmark-a-diagnostic-framework-for-planning-capabilities-in-llm.md", + "title": "\"Agent Planning Benchmark: A Diagnostic Framework for Planning Capabilities in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agent Planning Benchmark: A Diagnostic Framework for Planning Capabilities in LLM Agents\"", + "authors": "Haoyu Sun, Wenxuan Wang, Mingyang Song, Jujie He, Weinan Zhang, Yang Liu, Yang Yang, Yu Cheng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04874", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-04990-from-agent-traces-to-trust-a-survey-of-evidence-tracing-and-execution-provenance.md", + "title": "\"From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents\"", + "authors": "Yiqi Wang, Jiaqi Zhang, Taotao Cai, Zirui Liu, Qingqiang Sun, Zequn Sun, Zhangkai Wu, Manqing Dong, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.04990", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05241-search-time-contamination-in-deep-research-agents-measuring-performance-inflatio.md", + "title": "\"Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation\"", + "authors": "Yongjie Wang, Xinyue Zhang, Kunhong Yao, Zhiwei Zeng, Kaisong Song, Jun Lin, Zhiqi Shen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05241", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05263-policy-conditioned-counterfactual-credit-for-verifiable-reinforcement-learning-o.md", + "title": "Policy-Conditioned Counterfactual Credit for Verifiable Reinforcement Learning of Long-Horizon Language Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Policy-Conditioned Counterfactual Credit for Verifiable Reinforcement Learning of Long-Horizon Language Agents", + "authors": "Renwei Meng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05263", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05414-when-evidence-is-sparse-weakly-supervised-early-failure-alerting-in-dialogs-and-.md", + "title": "\"When Evidence is Sparse: Weakly Supervised Early Failure Alerting in Dialogs and LLM-Agent Trajectories\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When Evidence is Sparse: Weakly Supervised Early Failure Alerting in Dialogs and LLM-Agent Trajectories\"", + "authors": "Avinash Baidya, Xinran Liang, Ruocheng Guo, Xiang Gao, Kamalika Das", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05414", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.HC", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05436-ten-headache-specialists-versus-artificial-intelligence-for-clinical-literature-.md", + "title": "\"Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison\"", + "authors": "Alejandro Lozano, Keiko Ihara, Ping-Hao Yang, Carrie E. Robertson, Jennifer Stern, Allan Purdy, Hsiangkuo Yuan, Pengfei Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05436", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05463-psebench-a-controllable-and-verifiable-benchmark-for-evaluating-llms-in-patient-.md", + "title": "\"PSEBench: A Controllable and Verifiable Benchmark for Evaluating LLMs in Patient Safety Event Triage\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PSEBench: A Controllable and Verifiable Benchmark for Evaluating LLMs in Patient Safety Event Triage\"", + "authors": "Keqi Han, Ryan Young, Annabel Strauss, Lindsey Hughes, Katharine M. Nesbitt, Nicole Schueler, Che Ngufor, Carl Yang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05463", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05548-adk-arena-evaluating-agent-development-kits-via-llm-as-a-developer.md", + "title": "\"ADK Arena: Evaluating Agent Development Kits via LLM-as-a-Developer\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ADK Arena: Evaluating Agent Development Kits via LLM-as-a-Developer\"", + "authors": "Jintao Huang, Xiaomin Li, Gaurav Mittal, Yu Hu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05548", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation, autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05558-autoregressive-diffusion-world-models-for-off-policy-evaluation-of-llm-agents.md", + "title": "Autoregressive Diffusion World Models for Off-Policy Evaluation of LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Autoregressive Diffusion World Models for Off-Policy Evaluation of LLM Agents", + "authors": "Kaixuan Liu, Guojun Xiong, Weinan Zhang, Shengpu Tang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05558", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05622-adaplanbench-evaluating-adaptive-planning-in-large-language-model-agents-under-w.md", + "title": "\"AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints\"", + "authors": "Jiayu Liu, Cheng Qian, Zhenhailong Wang, Bingxuan Li, Jiateng Liu, Heng Wang, Jeonghwan Kim, Yumeng Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05622", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05658-agent-orchestrated-adaptive-rag-a-comparative-study-on-structured-and-multi-hop-.md", + "title": "\"Agent-Orchestrated Adaptive RAG: A Comparative Study on Structured and Multi-Hop Retrieval\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agent-Orchestrated Adaptive RAG: A Comparative Study on Structured and Multi-Hop Retrieval\"", + "authors": "Anuj Maharjan, Devinder Kaur, Richard Molyet", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05658", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05684-adamem-test-time-adaptive-memory-for-language-agents.md", + "title": "\"AdaMEM: Test-Time Adaptive Memory for Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AdaMEM: Test-Time Adaptive Memory for Language Agents\"", + "authors": "Yunxiang Zhang, Yiheng Li, Ali Payani, Lu Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05684", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory, language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05711-beyond-tokens-a-unified-framework-for-latent-communication-in-llm-based-multi-ag.md", + "title": "\"Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems\"", + "authors": "Yingzhuo Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05711", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-05805-from-risk-classification-to-action-plan-remediation-a-guardrail-feedback-driven-.md", + "title": "\"From Risk Classification to Action Plan Remediation: A Guardrail Feedback Driven Framework for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Risk Classification to Action Plan Remediation: A Guardrail Feedback Driven Framework for LLM Agents\"", + "authors": "Yuhao Sun, Jiacheng Zhang, Shaanan Cohney, Zhexin Zhang, Feng Liu, Xingliang Yuan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.05805", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-06054-beyond-similarity-trustworthy-memory-search-for-personal-ai-agents.md", + "title": "\"Beyond Similarity: Trustworthy Memory Search for Personal AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Similarity: Trustworthy Memory Search for Personal AI Agents\"", + "authors": "Jiawen Zhang, Kejia Chen, Jiachen Ma, Yangfan Hu, Lipeng He, Yechao Zhang, Jian Liu, Xiaohu Yang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.06054", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-06090-beyond-semantic-organization-memory-as-execution-state-management-for-long-horiz.md", + "title": "\"Beyond Semantic Organization: Memory as Execution State Management for Long-Horizon Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Semantic Organization: Memory as Execution State Management for Long-Horizon Agents\"", + "authors": "Yaoqi Chen, Haibin Lai, Yuru Feng, Chuyu Han, Qianxi Zhang, Baotong Lu, Menghao Li, Xinjiang Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.06090", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-06388-humans-almanac-a-human-collaboration-dataset-of-action-level-mental-model-annota.md", + "title": "\"Humans' ALMANAC: A Human Collaboration Dataset of Action-Level Mental Model Annotations for Agent Collaboration\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Humans' ALMANAC: A Human Collaboration Dataset of Action-Level Mental Model Annotations for Agent Collaboration\"", + "authors": "Jiaju Chen, Yuxuan Lu, Jiayi Su, Chaoran Chen, Songlin Xiao, Zheng Zhang, Yun Wang, Yunyao Li, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.06388", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-06399-collabsim-a-cscw-grounded-methodology-for-investigating-collaborative-competence.md", + "title": "\"CollabSim: A CSCW-Grounded Methodology for Investigating Collaborative Competence of LLM Agents through Controlled Multi-Agent Experiments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CollabSim: A CSCW-Grounded Methodology for Investigating Collaborative Competence of LLM Agents through Controlled Multi-Agent Experiments\"", + "authors": "Jiaju Chen, Bo Sun, Yuxuan Lu, Yun Wang, Dakuo Wang, Bingsheng Yao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.06399", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-06448-agent-memory-characterization-and-system-implications-of-stateful-long-horizon-w.md", + "title": "\"Agent Memory: Characterization and System Implications of Stateful Long-Horizon Workloads\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agent Memory: Characterization and System Implications of Stateful Long-Horizon Workloads\"", + "authors": "Yasmine Omri, Ziyu Gan, Zachary Broveak, Robin Geens, Zexue He, Alex Pentland, Marian Verhelst, Tsachy Weissman, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.06448", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-06462-benchmark-everything-everywhere-all-at-once.md", + "title": "Benchmark Everything Everywhere All at Once", + "type": "paper", + "meta": { + "type": "paper", + "title": "Benchmark Everything Everywhere All at Once", + "authors": "Shiyun Xiong, Dongming Wu, Peiwen Sun, Yuang Ai, Bokang Yang, Wencheng Han, Xiao-Hui Li, Xiangyu Yue", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.06462", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-06473-mlevolve-a-self-evolving-framework-for-automated-machine-learning-algorithm-disc.md", + "title": "\"MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery\"", + "authors": "Shangheng Du, Xiangchao Yan, Jinxin Shi, Zongsheng Cao, Shiyang Feng, Zichen Liang, Boyuan Sun, Tianshuo Peng, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.06473", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-04", + "updated_at": "2026-06-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "multi-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-07314-qbuglm-an-agentic-benchmarking-framework-for-llm-based-quantum-software-debuggin.md", + "title": "\"QBugLM: An Agentic Benchmarking Framework for LLM-based Quantum Software Debugging\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"QBugLM: An Agentic Benchmarking Framework for LLM-based Quantum Software Debugging\"", + "authors": "An B. B. Pham, Hoa T. Nguyen, Muhammad Usman", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.07314", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-05", + "updated_at": "2026-06-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "reasoning", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.ET", + "quant-ph" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-07379-do-coding-agents-deceive-us-detecting-and-preventing-cheating-via-capped-evaluat.md", + "title": "Do Coding Agents Deceive Us? Detecting and Preventing Cheating via Capped Evaluation with Randomized Tests", + "type": "paper", + "meta": { + "type": "paper", + "title": "Do Coding Agents Deceive Us? Detecting and Preventing Cheating via Capped Evaluation with Randomized Tests", + "authors": "Thanawat Lodkaew, Johannes Ackermann, Soichiro Nishimori, Nontawat Charoenphakdee, Masashi Sugiyama, Takashi Ishida", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.07379", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-05", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL", + "stat.ME" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-07402-m-3-exam-benchmarking-multimodal-memory-for-realistic-user-agent-interactions.md", + "title": "\"M$^3$Exam: Benchmarking Multimodal Memory for Realistic User-Agent Interactions\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"M$^3$Exam: Benchmarking Multimodal Memory for Realistic User-Agent Interactions\"", + "authors": "Zhengjun Huang, Wenxuan Liu, Zhoujin Tian, Wei Chen, Junle Chen, Yuqian Wu, Fangyuan Zhang, Qintian Guo, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.07402", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-05", + "updated_at": "2026-06-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-07591-researchclawbench-a-benchmark-for-end-to-end-autonomous-scientific-research.md", + "title": "\"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research\"", + "authors": "Wanghan Xu, Shuo Li, Tianlin Ye, Qinglong Cao, Yixin Chen, Hengjian Gao, Yiheng Wang, Qi Li, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.07591", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-07595-visualleakbench-reproducible-action-boundary-propagation-failures-in-vision-lang.md", + "title": "\"VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents\"", + "authors": "Youting Wang, Yuan Tang, Yitian Qian, Chen Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.07595", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-29", + "updated_at": "2026-05-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-07682-swe-marathon-can-agents-autonomously-complete-ultra-long-horizon-software-work.md", + "title": "\"SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work?\"", + "authors": "Rishi Desai, Jesse Hu, Joan Cabezas, Neel Harsola, Pratyush Shukla, Roey Ben Chaim, Adnan El Assadi, Omkaar Mukund Kamath, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.07682", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-05", + "updated_at": "2026-06-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-07711-rosetta-memory-adaptive-memory-for-cross-llm-agents.md", + "title": "\"Rosetta Memory: Adaptive Memory for Cross-LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Rosetta Memory: Adaptive Memory for Cross-LLM Agents\"", + "authors": "Hao Yang, Shiqi Shen, Haoxuan Li, Zhipeng Wang, Zhi Gong, Xu Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.07711", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-05", + "updated_at": "2026-06-05", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "memory", + "planning", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-07836-agentic-multi-fidelity-learning-of-quasiparticle-and-excitonic-properties.md", + "title": "Agentic multi-fidelity learning of quasiparticle and excitonic properties", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agentic multi-fidelity learning of quasiparticle and excitonic properties", + "authors": "Arnab Neogi, Aaron Forde, Christopher A. Lane, Sergei Tretiak, Jian-Xin Zhu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.07836", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-05", + "updated_at": "2026-06-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cond-mat.mtrl-sci", + "cond-mat.stat-mech", + "cs.AI", + "physics.comp-ph", + "quant-ph" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-07867-the-cold-start-safety-gap-in-llm-agents.md", + "title": "The Cold-Start Safety Gap in LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "The Cold-Start Safety Gap in LLM Agents", + "authors": "Chung-En Sun, Linbo Liu, Tsui-Wei Weng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.07867", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-05", + "updated_at": "2026-06-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-08162-silent-failure-in-llm-agent-systems-the-entropy-principle-and-the-inevitable-dis.md", + "title": "\"Silent Failure in LLM Agent Systems: The Entropy Principle and the Inevitable Disorder of Autonomous Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Silent Failure in LLM Agent Systems: The Entropy Principle and the Inevitable Disorder of Autonomous Agents\"", + "authors": "Dexing Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.08162", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-06", + "updated_at": "2026-06-06", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-08172-the-governance-of-human-llm-interaction-safety-gating-civility-steering-and-affe.md", + "title": "\"The Governance of Human-LLM Interaction: Safety Gating, Civility Steering, and Affective Default Lock-In\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Governance of Human-LLM Interaction: Safety Gating, Civility Steering, and Affective Default Lock-In\"", + "authors": "Manuele Reani, Hongjian Zhang, Hongyu Tian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.08172", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-06", + "updated_at": "2026-06-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.HC", + "cs.AI", + "cs.CY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-08274-toward-human-centered-multi-agent-systems-integrating-cognition-culture-values-a.md", + "title": "\"Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents\"", + "authors": "Safia Baloch, Rahemeen Khan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.08274", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-06", + "updated_at": "2026-06-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "24", + "collection_queries": "autonomous-agent-llm, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-08340-benchmarking-open-ended-multi-agent-coordination-in-language-agents.md", + "title": "Benchmarking Open-Ended Multi-Agent Coordination in Language Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Benchmarking Open-Ended Multi-Agent Coordination in Language Agents", + "authors": "Kale-ab Abebe Tessera, Andras Szecsenyi, Cameron Barker, Alexander Rutherford, Davide Paglieri, Aidan Scannell, Henry Gouk, Elliot J. Crowley, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.08340", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-06", + "updated_at": "2026-06-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "28", + "collection_queries": "autonomous-agent-llm, language-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-08531-vesta-a-fully-automated-scenario-generation-and-safety-evaluation-framework-for-.md", + "title": "\"VESTA: A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VESTA: A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents\"", + "authors": "Lu Jia, Haibo Tong, Feifei Zhao, Jindong Li, Dongqi Liang, Ping Wu, Qian Zhang, Yi Zeng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.08531", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-07", + "updated_at": "2026-06-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-08625-from-holistic-evaluation-to-structured-criteria-rubrics-across-the-evolving-llm-.md", + "title": "\"From Holistic Evaluation to Structured Criteria: Rubrics Across the Evolving LLM Landscape\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Holistic Evaluation to Structured Criteria: Rubrics Across the Evolving LLM Landscape\"", + "authors": "Hao Chen, Ziyu Han, Yukun Yan, Qingfu Zhu, Maosong Sun, Wanxiang Che", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.08625", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-07", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-08790-rails-verification-native-clearing-for-agentic-commerce.md", + "title": "\"RAILS: Verification-Native Clearing For Agentic Commerce\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RAILS: Verification-Native Clearing For Agentic Commerce\"", + "authors": "Adrian de Valois-Franklin, Alex Bogdan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.08790", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-07", + "updated_at": "2026-06-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CR", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-08960-hardening-agent-benchmarks-with-adversarial-hacker-fixer-loops.md", + "title": "Hardening Agent Benchmarks with Adversarial Hacker-Fixer Loops", + "type": "paper", + "meta": { + "type": "paper", + "title": "Hardening Agent Benchmarks with Adversarial Hacker-Fixer Loops", + "authors": "Ziqian Zhong, Ivgeni Segal, Ivan Bercovich, Shashwat Saxena, Kexun Zhang, Aditi Raghunathan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.08960", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09037-a-multi-agent-system-for-ipmsm-design-optimization-via-an-fea-ai-hybrid-approach.md", + "title": "A Multi-Agent System for IPMSM Design Optimization via an FEA-AI Hybrid Approach", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Multi-Agent System for IPMSM Design Optimization via an FEA-AI Hybrid Approach", + "authors": "Jinseong Han, Sunwoong Yang, Namwoo Kang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09037", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09071-reflect-intervention-supported-error-attribution-for-silent-failures-in-llm-agen.md", + "title": "\"REFLECT: Intervention-Supported Error Attribution for Silent Failures in LLM Agent Traces\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"REFLECT: Intervention-Supported Error Attribution for Silent Failures in LLM Agent Traces\"", + "authors": "Xiaofeng Lin, Yingxu Wang, Tung Sum Thomas Kwok, Daniel Guo, Sahil Arun Nale, Charles Fleming, Guang Cheng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09071", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09198-mass-deep-research-for-social-sciences-with-memory-augmented-social-simulation.md", + "title": "\"MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation\"", + "authors": "Yongrui Liu, Deyi Xiong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09198", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09316-anything2skill-compiling-external-knowledge-into-reusable-skills-for-agents.md", + "title": "\"Anything2Skill: Compiling External Knowledge into Reusable Skills for Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Anything2Skill: Compiling External Knowledge into Reusable Skills for Agents\"", + "authors": "Qianjun Pan, Yutao Yang, Junsong Li, Jie Zhou, Kai Chen, Xin Li, Qin Chen, Liang He", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09316", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09399-runagent-superbrowser-a-theory-of-autonomous-web-navigation-grounded-in-human-br.md", + "title": "\"RunAgent SuperBrowser: A Theory of Autonomous Web Navigation Grounded in Human Browsing Behaviour\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RunAgent SuperBrowser: A Theory of Autonomous Web Navigation Grounded in Human Browsing Behaviour\"", + "authors": "Radeen Mostafa, Sawradip Saha", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09399", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09426-weavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybr.md", + "title": "\"WeaveBench: A Long-Horizon, Real-World Benchmark for Computer-Use Agents with Hybrid Interfaces\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"WeaveBench: A Long-Horizon, Real-World Benchmark for Computer-Use Agents with Hybrid Interfaces\"", + "authors": "Wanli Li, Bowen Zhou, Yunyao Yu, Zhou Xu, Yifan Yang, Dongsheng Li, Caihua Shan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09426", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09447-aliyunconsoleagent-training-web-agents-in-real-world-cloud-environments-via-dist.md", + "title": "\"AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning\"", + "authors": "Bojie Rong, Zheyu Shen, Qiaoping Wang, Pengfei Kang, Yang Xu, Yawen Wei, Hanyu Wu, Zhi Zhao, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09447", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09483-memory-beyond-recall-a-dual-process-cognitive-memory-system-for-self-evolving-ll.md", + "title": "\"Memory Beyond Recall: A Dual-Process Cognitive Memory System for Self-Evolving LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Memory Beyond Recall: A Dual-Process Cognitive Memory System for Self-Evolving LLM Agents\"", + "authors": "Tianxiang Fei, Mingyang Song, Mao Zheng, Xiang Yu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09483", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09549-secureclaw-clawing-back-control-of-llm-agents.md", + "title": "\"SecureClaw: Clawing Back Control of LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SecureClaw: Clawing Back Control of LLM Agents\"", + "authors": "Yuhan Ma, Stefan Schmid", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09549", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09738-hdsl-a-hierarchical-domain-specific-language-for-structured-3d-indoor-scene-gene.md", + "title": "\"HDSL: A Hierarchical Domain-Specific Language for Structured 3D Indoor Scene Generation and Localized Editing with LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HDSL: A Hierarchical Domain-Specific Language for Structured 3D Indoor Scene Generation and Localized Editing with LLM Agents\"", + "authors": "Letian Li, Chao Shen, Shuzhao Xie, Chenghao Gu, ZhengXiao He, Yu Meng, Xin Yang, Wenyuan Jiang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09738", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09764-iosworld-a-benchmark-for-personally-intelligent-phone-agents.md", + "title": "\"iOSWorld: A Benchmark for Personally Intelligent Phone Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"iOSWorld: A Benchmark for Personally Intelligent Phone Agents\"", + "authors": "Lawrence Keunho Jang, Mareks Woodside, Geronimo Carom, Andrew Keunwoo Jang, Jing Yu Koh, Ruslan Salakhutdinov", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09764", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09774-auto-configuring-scientific-simulators-with-lightweight-coding-agent-adapters.md", + "title": "Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters", + "type": "paper", + "meta": { + "type": "paper", + "title": "Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters", + "authors": "Matthew Ho, Brian Liu, Jixuan Chen, Audrey Wang, Lianhui Qin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09774", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09863-from-confident-closing-to-silent-failure-characterizing-false-success-in-llm-age.md", + "title": "\"From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents\"", + "authors": "Laksh Advani", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09863", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-01", + "updated_at": "2026-06-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-09961-3spo-state-score-supervised-policy-optimization-for-llm-agents.md", + "title": "\"3SPO: State-Score-Supervised Policy Optimization for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"3SPO: State-Score-Supervised Policy Optimization for LLM Agents\"", + "authors": "Yu Han, Kailing Li, Yang Jiao, Yulin Dai, Yuqian Fu, Linhai Zhuo, Tianwen Qian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.09961", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10209-less-context-better-agents-efficient-context-engineering-for-long-horizon-tool-u.md", + "title": "\"Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents\"", + "authors": "Abhilasha Lodha, Mahsa Pahlavikhah Varnosfaderani, Abir Chakraborty, Abhinav Mithal", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10209", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-08", + "updated_at": "2026-06-08", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10304-mirage-a-polarity-flipping-encoding-subspace-in-llm-agents.md", + "title": "\"MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents\"", + "authors": "Pratibha Revankar, Kargi Chauhan, Jihye Kim, Sadiba Nusrat Nur, Vincent Siu, Chenguang Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10304", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10316-tabclaw-an-interactive-and-self-evolving-agent-for-spreadsheet-manipulation-and-.md", + "title": "\"TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning\"", + "authors": "Mingyue Cheng, Shuo Yu, Daoyu Wang, Qingchuan Li, Xiaoyu Tao, Qingyang Mao, Yitong Zhou, Qi Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10316", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10381-agentic-hybrid-rag-for-evidence-grounded-muon-collider-analysis.md", + "title": "Agentic Hybrid RAG for Evidence-Grounded Muon Collider Analysis", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agentic Hybrid RAG for Evidence-Grounded Muon Collider Analysis", + "authors": "Ruobing Jiang, Dawei Fu, Cheng Jiang, Tianyi Yang, Zijian Wang, Youpeng Wu, Yong Ban, Yajun Mao, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10381", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "hep-ex", + "cs.AI", + "cs.CL", + "cs.IR", + "physics.ins-det" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10394-stage-claw-automated-state-based-agent-benchmarking-for-realistic-scenarios.md", + "title": "\"STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios\"", + "authors": "Sirui Liang, Bohan Yu, Peiyu Wang, Shiguang Guo, Wenxing Hu, Pengfei Cao, Jian Zhao, Cao Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10394", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10423-webchallenger-a-reliable-and-efficient-generalist-web-agent.md", + "title": "\"WebChallenger: A Reliable and Efficient Generalist Web Agent\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"WebChallenger: A Reliable and Efficient Generalist Web Agent\"", + "authors": "Jayoo Hwang, Xiaowen Zhang, Vedant Padwal", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10423", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "embodied-agent", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10507-hipif-hierarchical-planning-and-information-folding-for-long-horizon-llm-agent-l.md", + "title": "\"HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning\"", + "authors": "Juncheng Diao, Zhicong Lu, Peiguang Li, Yongwei Zhou, Changyuan Tian, Qingbin Li, Rongxiang Weng, Jingang Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10507", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation, autonomous-agent-llm, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10532-activemem-distributed-active-memory-for-long-horizon-llm-reasoning.md", + "title": "\"ActiveMem: Distributed Active Memory for Long-Horizon LLM Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ActiveMem: Distributed Active Memory for Long-Horizon LLM Reasoning\"", + "authors": "Yunhan Jiang, Wenbin Duan, Shasha Guo, Liang Pang, Xiaoqian Sun, Huawei Shen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10532", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10577-agenticnav-zero-shot-vision-and-language-navigation-as-a-tool-calling-harness.md", + "title": "\"AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness\"", + "authors": "Yijian Li, Changze Li, Hantian Shi, Jiaying Luo, Jiyuan Cai, Ming Yang, Tong Qin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10577", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10616-learning-what-to-remember-observability-safe-memory-retention-via-constrained-op.md", + "title": "\"Learning What to Remember: Observability-Safe Memory Retention via Constrained Optimization for Long-Horizon Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Learning What to Remember: Observability-Safe Memory Retention via Constrained Optimization for Long-Horizon Language Agents\"", + "authors": "Qingcan Kang, Liu Mingyang, Shixiong Kai, Kaichao Liang, Tao Zhong, Mingxuan Yuan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10616", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10677-infini-memory-maintainable-topic-documents-for-long-term-llm-agent-memory.md", + "title": "\"Infini Memory: Maintainable Topic Documents for Long-Term LLM Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Infini Memory: Maintainable Topic Documents for Long-Term LLM Agent Memory\"", + "authors": "Suozhao Ji, Baodong Wu, Zehao Wang, Lei Xia, Qingping Li, Ruisong Wang, Wenbo Ding, Zhenhua Zhu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10677", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10684-divide-and-cooperate-role-decomposed-multi-agent-llm-training-with-cross-agent-l.md", + "title": "\"Divide and Cooperate: Role-Decomposed Multi-Agent LLM Training with Cross-Agent Learning Signals\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Divide and Cooperate: Role-Decomposed Multi-Agent LLM Training with Cross-Agent Learning Signals\"", + "authors": "Jaewan Park, Solbee Cho, Jay-Yoon Lee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10684", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10742-memvenom-triggered-poisoning-of-multimodal-memories-in-web-agents.md", + "title": "\"MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents\"", + "authors": "Yv Zhang, Hao Sun, Hao Fang, Kuofeng Gao, Fan Mo, Bin Chen, Shu-Tao Xia, Yaowei Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10742", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10749-toward-secure-llm-agents-threat-surfaces-attacks-defenses-and-evaluation.md", + "title": "\"Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation\"", + "authors": "Yuchen Ling, Shengcheng Yu, Zhenyu Chen, Chunrong Fang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10749", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "25", + "collection_queries": "agent-safety, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10921-trace-only-what-you-need-structure-aware-on-demand-hypergraph-memory-for-long-do.md", + "title": "\"Trace Only What You Need: Structure-Aware On-Demand Hypergraph Memory for Long-Document Question Answering\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Trace Only What You Need: Structure-Aware On-Demand Hypergraph Memory for Long-Document Question Answering\"", + "authors": "Xiangjun Zai, Xingyu Tan, Chen Chen, Xiaoyang Wang, Wenjie Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10921", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-10933-frontier-coding-agents-use-metaprogramming-to-adapt-to-unfamiliar-programming-la.md", + "title": "Frontier Coding Agents Use Metaprogramming to Adapt to Unfamiliar Programming Languages", + "type": "paper", + "meta": { + "type": "paper", + "title": "Frontier Coding Agents Use Metaprogramming to Adapt to Unfamiliar Programming Languages", + "authors": "Aman Sharma, Sushrut Thorat, Paras Chopra", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.10933", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11042-workflow-gym-towards-long-horizon-evaluation-of-computer-use-agentic-tasks-in-re.md", + "title": "\"Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields\"", + "authors": "Liya Zhu, Jingzhe Ding, Jian Zhang, Jianbo Xue, Shihao Liang, Ge Zhang, Yi Zhu, Duju Zeng, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11042", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11078-a-history-aware-visually-grounded-critic-for-computer-use-agents.md", + "title": "A History-Aware Visually Grounded Critic for Computer Use Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "A History-Aware Visually Grounded Critic for Computer Use Agents", + "authors": "Jaewoo Lee, Zaid Khan, Archiki Prasad, Justin Chih-Yao Chen, Supriyo Chakraborty, Kartik Balasubramaniam, Sambit Sahu, Elias Stengel-Eskin, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11078", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11079-vista-a-versatile-interactive-user-simulation-toolkit-for-agent-evaluation.md", + "title": "\"VISTA: A Versatile Interactive User Simulation Toolkit for Agent Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VISTA: A Versatile Interactive User Simulation Toolkit for Agent Evaluation\"", + "authors": "Yunan Lu, Ryan Shea, Yusen Zhang, Zhou Yu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11079", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11119-trace-a-unified-rollout-budget-allocation-framework-for-efficient-agentic-reinfo.md", + "title": "\"TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning\"", + "authors": "Heming Zou, Qi Wang, Yun Qu, Yuhang Jiang, Lizhou Cai, Yixiu Mao, Ru Peng, Xin Xu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11119", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11176-data-journalist-agent-transforming-data-into-verifiable-multimodal-stories.md", + "title": "\"Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories\"", + "authors": "Kevin Qinghong Lin, Batu EI, Yuhong Shi, Pan Lu, Philip Torr, James Zou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11176", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.CL", + "cs.CY", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11349-knowing-when-to-ask-self-gated-clarification-for-hierarchical-language-agents.md", + "title": "\"Knowing When to Ask: Self-Gated Clarification for Hierarchical Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Knowing When to Ask: Self-Gated Clarification for Hierarchical Language Agents\"", + "authors": "Aijing Gao, Yiming Kang, Mengdie Flora Wang, Jae Oh Woo", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11349", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11354-a-zero-shot-multi-agent-framework-for-human-building-interaction-via-programmati.md", + "title": "A Zero-Shot Multi-Agent Framework for Human-Building Interaction via Programmatic Reasoning", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Zero-Shot Multi-Agent Framework for Human-Building Interaction via Programmatic Reasoning", + "authors": "Yuqi Wang, Gulai Shen, Ali Mehmani", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11354", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-09", + "updated_at": "2026-06-09", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.ET" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11680-organize-then-retrieve-hierarchical-memory-navigation-for-efficient-agents.md", + "title": "\"Organize then Retrieve: Hierarchical Memory Navigation for Efficient Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Organize then Retrieve: Hierarchical Memory Navigation for Efficient Agents\"", + "authors": "Hao-Lun Hsu, Nikki Lijing Kuang, Boyi Liu, Zhewei Yao, Yuxiong He", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11680", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11688-goal-autopilot-a-verifiable-anti-fabrication-firewall-for-unattended-long-horizo.md", + "title": "\"Goal-Autopilot: A Verifiable Anti-Fabrication Firewall for Unattended Long-Horizon Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Goal-Autopilot: A Verifiable Anti-Fabrication Firewall for Unattended Long-Horizon Agents\"", + "authors": "Youwang Deng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11688", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11702-medcta-a-benchmark-for-clinical-tool-agents.md", + "title": "\"MedCTA: A Benchmark for Clinical Tool Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MedCTA: A Benchmark for Clinical Tool Agents\"", + "authors": "Tajamul Ashraf, Hyewon Jeong, Fida Mohammad Thoker, Bernard Ghanem", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11702", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-11869-agents-all-the-way-down-a-methodology-for-building-custom-ai-agents-from-substra.md", + "title": "Agents All the Way Down; A Methodology for Building Custom AI Agents from Substrate to Production", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agents All the Way Down; A Methodology for Building Custom AI Agents from Substrate to Production", + "authors": "Marc Alier Forment, Juanan Pereira, Francisco José García-Peñalvo, María José Casañ Guerrero", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.11869", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12195-internvideo3-agentify-foundation-models-with-multimodal-contextual-reasoning.md", + "title": "\"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning\"", + "authors": "Ziang Yan, Sheng Xia, Jiashuo Yu, Yue Wu, Tianxiang Jiang, Songze Li, Kanghui Tian, Yicheng Xu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12195", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12320-a-five-plane-reference-architecture-for-runtime-governance-of-production-ai-agen.md", + "title": "A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents", + "authors": "Krti Tallam", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12320", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CC", + "cs.CR", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12341-ocelot-inference-leakage-budgets-for-privacy-preserving-llm-agents.md", + "title": "\"OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents\"", + "authors": "Jin Xie, Songze Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12341", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12344-claw-swe-bench-a-benchmark-for-evaluating-openclaw-style-agent-harnesses-on-codi.md", + "title": "\"Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks\"", + "authors": "Mengyu Zheng, Kai Han, Boxun Li, Haiyang Xu, Yuchuan Tian, Wei He, Hang Zhou, Jianyuan Guo, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12344", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12384-appo-agentic-procedural-policy-optimization.md", + "title": "\"APPO: Agentic Procedural Policy Optimization\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"APPO: Agentic Procedural Policy Optimization\"", + "authors": "Xucong Wang, Ziyu Ma, Yong Wang, Yuxiang Ji, Shidong Yang, Guanhua Chen, Pengkun Wang, Xiangxiang Chu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12384", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12563-arbor-tree-search-as-a-cognition-layer-for-autonomous-agents.md", + "title": "\"Arbor: Tree Search as a Cognition Layer for Autonomous Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Arbor: Tree Search as a Cognition Layer for Autonomous Agents\"", + "authors": "Neha Prakriya, Chaojun Hou, Zheng Gong, Huasha Zhao, Xi Zhao, Mou Li, Zhenyu Gu, Emad Barsoum", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12563", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12586-beyond-attack-success-rate-examining-trigger-leakage-in-vision-language-agentic-.md", + "title": "\"Beyond Attack Success Rate: Examining Trigger Leakage in Vision-Language Agentic Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Attack Success Rate: Examining Trigger Leakage in Vision-Language Agentic Systems\"", + "authors": "Jiamin Chang, Salil Kanhere, Piotr Koniusz, Jason, Xue, Hammond Pearce", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12586", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12634-keep-policy-gradient-in-charge-sibling-guided-credit-distillation-for-long-horiz.md", + "title": "\"Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents\"", + "authors": "Tianyu Ding, Jianhong Xin, Juan Pablo De la Cruz Weinstein", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12634", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12657-trajgenagent-a-hierarchical-llm-agent-for-human-mobility-trajectory-generation.md", + "title": "\"TrajGenAgent: A Hierarchical LLM Agent for Human Mobility Trajectory Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TrajGenAgent: A Hierarchical LLM Agent for Human Mobility Trajectory Generation\"", + "authors": "Siyu Li, Toan Tran, Lingyi Zhao, Khurram Shafique, Li Xiong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12657", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.DB", + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12674-evoflux-inference-time-evolution-of-executable-tool-workflows-for-compact-agents.md", + "title": "\"Evoflux: Inference-Time Evolution of Executable Tool Workflows for Compact Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Evoflux: Inference-Time Evolution of Executable Tool Workflows for Compact Agents\"", + "authors": "Kushal Raj Bhandari, Ling Yue, Ching-Yun Ko, Dhaval Patel, Shaowu Pan, Pin-Yu Chen, Jianxi Gao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12674", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "function-calling, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12703-smsr-certified-defence-against-runtime-memory-poisoning-in-persistent-llm-agent-.md", + "title": "\"SMSR: Certified Defence Against Runtime Memory Poisoning in Persistent LLM Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SMSR: Certified Defence Against Runtime Memory Poisoning in Persistent LLM Agent Systems\"", + "authors": "Tarun Sharma", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12703", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12780-proplay-procedural-world-models-for-self-evolving-llm-agents.md", + "title": "\"ProPlay: Procedural World Models for Self-Evolving LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ProPlay: Procedural World Models for Self-Evolving LLM Agents\"", + "authors": "Yijun Ma, Zehong Wang, Yiyang Li, Ziming Li, Xiaoguang Guo, Weixiang Sun, Chuxu Zhang, Yanfang Ye", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12780", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12837-lohosearch-benchmarking-long-horizon-search-agents-beyond-the-human-difficulty-c.md", + "title": "\"LoHoSearch: Benchmarking Long-Horizon Search Agents Beyond the Human Difficulty Ceiling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LoHoSearch: Benchmarking Long-Horizon Search Agents Beyond the Human Difficulty Ceiling\"", + "authors": "Jiarui Zhao, Rongzhi Zhang, Lingchuan Liu, Hao Yang, Xunliang Cai, Xi Su", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12837", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-12945-learning-what-to-remember-a-cognitively-grounded-multi-factor-value-model-for-ag.md", + "title": "\"Learning What to Remember: A Cognitively Grounded Multi-Factor Value Model for Agentic Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Learning What to Remember: A Cognitively Grounded Multi-Factor Value Model for Agentic Memory\"", + "authors": "Zhibao Chen, Qian Cheng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.12945", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13148-terrabench-can-agents-reason-over-heterogeneous-earth-system-data.md", + "title": "\"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?\"", + "authors": "Dat Tien Nguyen, Thao Nguyen, Fadillah Adamsyah Maani, Huy M. Le, Muhammad Umer Sheikh, Numan Saeed, Muhammad Haris Khan, Salman Khan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13148", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13177-memrefine-llm-guided-compression-for-long-term-agent-memory.md", + "title": "\"MemRefine: LLM-Guided Compression for Long-Term Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemRefine: LLM-Guided Compression for Long-Term Agent Memory\"", + "authors": "Minjae Kim, Jinheon Baek, Soyeong Jeong, Sung Ju Hwang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13177", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13192-reasoning-for-mobile-user-experience-with-multimodal-llms-task-benchmark-and-app.md", + "title": "\"Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach\"", + "authors": "Ruichao Mao, Zhou Fang, Teng Guo, Hao Yang, Yaping Li, Shaohua Peng, Maji Huang, Xiaoyu Lin, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13192", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13317-skillcat-contrastive-assessment-and-topology-aware-skill-self-evolution-for-llm-.md", + "title": "\"SkillCAT: Contrastive Assessment and Topology-Aware Skill Self-Evolution for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SkillCAT: Contrastive Assessment and Topology-Aware Skill Self-Evolution for LLM Agents\"", + "authors": "Kunfeng Chen, Qihuang Zhong, Juhua Liu, Bo Du", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13317", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13385-who-pays-the-price-stakeholder-centric-prompt-injection-benchmarking-for-real-wo.md", + "title": "Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents", + "authors": "Zihao Wang, Yiming Li, Yutong Wu, Zheyu Liu, Kangjie Chen, Fok Kar Wai, Pin-Yu Chen, Vrizlynn L. L. Thing, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13385", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.CY", + "cs.HC", + "cs.MM" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13602-epibench-verifiable-evaluation-of-ai-agents-on-epigenomics-analysis.md", + "title": "\"EpiBench: Verifiable Evaluation of AI Agents on Epigenomics Analysis\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EpiBench: Verifiable Evaluation of AI Agents on Epigenomics Analysis\"", + "authors": "Harihara Muralidharan, Reema Baskar, Soo Hee Lee, Tim Proctor, Kenny Workman", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13602", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13608-agentbeats-agentifying-agent-assessment-for-openness-standardization-and-reprodu.md", + "title": "\"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility\"", + "authors": "Xiaoyuan Liu, Jianhong Tu, Yuqi Chen, Siyuan Xie, Sihan Ren, Tianneng Shi, Gal Gantar, Evan Sandoval, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13608", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13643-recursive-agent-harnesses.md", + "title": "Recursive Agent Harnesses", + "type": "paper", + "meta": { + "type": "paper", + "title": "Recursive Agent Harnesses", + "authors": "Elias Lumer, Sahil Sen, Kevin Paul, Vamse Kumar Subbiah", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13643", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13663-hypertool-beyond-step-wise-tool-calls-for-tool-augmented-agents.md", + "title": "\"HyperTool: Beyond Step-Wise Tool Calls for Tool-Augmented Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HyperTool: Beyond Step-Wise Tool Calls for Tool-Augmented Agents\"", + "authors": "Yaxin Du, Yifan Zhou, Yujie Ge, Jiajun Wang, Xianghe Pang, Shuo Tang, Tuney Zheng, Bryan Dai, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13663", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13686-benchmarking-web-agent-safety-under-e-commerce-deceptive-interfaces.md", + "title": "Benchmarking Web Agent Safety under E-commerce Deceptive Interfaces", + "type": "paper", + "meta": { + "type": "paper", + "title": "Benchmarking Web Agent Safety under E-commerce Deceptive Interfaces", + "authors": "Zijing Shi, Meng Fang, Ling Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13686", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-26", + "updated_at": "2026-04-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.CY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13904-sana-what-matters-for-qa-agents-over-massive-data-lakes.md", + "title": "\"SANA: What Matters for QA Agents over Massive Data Lakes?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SANA: What Matters for QA Agents over Massive Data Lakes?\"", + "authors": "Austin Senna Wijaya, Jiaxiang Liu, Haonan Wang, Eugene Wu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13904", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.DB" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-13994-hidden-in-plain-sight-benchmarking-agent-safety-against-decomposition-attacks-wi.md", + "title": "\"Hidden in Plain Sight: Benchmarking Agent Safety Against Decomposition Attacks with DECOMPBENCH\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Hidden in Plain Sight: Benchmarking Agent Safety Against Decomposition Attacks with DECOMPBENCH\"", + "authors": "Vikhyath Kothamasu, Virginia Smith, Chhavi Yadav", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.13994", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-12", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "agent-safety, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-14106-naive-visual-memory-is-not-enough-a-failure-mode-study-of-gui-agents.md", + "title": "\"Naive Visual Memory is Not Enough: A Failure-Mode Study of GUI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Naive Visual Memory is Not Enough: A Failure-Mode Study of GUI Agents\"", + "authors": "Seoyoung Choi, Minseok Ko, Hyunseok Lee, Kunwoong Kim, Woomin Song, Chanseok Jeon, Jinwoo Shin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.14106", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-12", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-14470-gitofthoughts-version-controlled-reasoning-and-agent-memory-you-can-replay-diff-.md", + "title": "\"GitOfThoughts: Version-Controlled Reasoning and Agent Memory You Can Replay, Diff, and Merge\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GitOfThoughts: Version-Controlled Reasoning and Agent Memory You Can Replay, Diff, and Merge\"", + "authors": "Pavan C Shekar, Abhishek H S, Aswanth Krishnan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.14470", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-12", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-14502-from-chatbot-to-digital-colleague-the-paradigm-shift-toward-persistent-autonomou.md", + "title": "\"From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI\"", + "authors": "Yongheng Zhang, Ziang Liu, Jiaxuan Zhu, Shuai Wang, Xiangqi Chen, Haojing Huang, Jiayi Kuang, Siyu Chen, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.14502", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-12", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-14517-from-shield-to-target-denial-of-service-attacks-on-llm-based-agent-guardrails.md", + "title": "\"From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails\"", + "authors": "Yuguang Zhou, Xunguang Wang, Pingchuan Ma, Zhantong Xue, Zhaoyu Wang, Shuai Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.14517", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-12", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation, autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-14571-streammembench-streaming-evaluation-of-agent-memory-for-future-oriented-assistan.md", + "title": "\"StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance\"", + "authors": "Guanming Liu, Yuqi Ren, Hansu Gu, Peng Zhang, Weihang Wang, Jiahao Liu, Ning Gu, Tun Lu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.14571", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-12", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-14574-simmer-benchmarking-latent-failures-in-llm-executable-planning-with-a-world-mode.md", + "title": "\"SIMMER: Benchmarking Latent Failures in LLM Executable Planning with a World Model\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SIMMER: Benchmarking Latent Failures in LLM Executable Planning with a World Model\"", + "authors": "Xiaoxin Lu, Ranran Haoran Zhang, Rui Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.14574", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-12", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-14790-xflow-an-executable-protocol-programming-system-for-reliable-multi-agent-workflo.md", + "title": "\"XFlow: An Executable Protocol Programming System for Reliable Multi-Agent Workflows\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"XFlow: An Executable Protocol Programming System for Reliable Multi-Agent Workflows\"", + "authors": "Hanqi Li, Jing Peng, Zijian Wang, Lu Chen, Kai Yu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.14790", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "memory", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.PL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-14805-knowledge-based-zero-replay-debugging-of-multi-agent-llm-traces.md", + "title": "Knowledge-Based Zero-Replay Debugging of Multi-Agent LLM Traces", + "type": "paper", + "meta": { + "type": "paper", + "title": "Knowledge-Based Zero-Replay Debugging of Multi-Agent LLM Traces", + "authors": "Dong Ho Kang, Hyeonjeong Cha, Daein Weon", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.14805", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15017-are-online-skill-and-memory-modules-always-worth-their-tokens-a-budget-constrain.md", + "title": "Are Online Skill and Memory Modules Always Worth Their Tokens? A Budget-Constrained Study of Web Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Are Online Skill and Memory Modules Always Worth Their Tokens? A Budget-Constrained Study of Web Agents", + "authors": "Sina Hajimiri, Masih Aminbeidokhti, Jose Dolz, Ismail Ben Ayed, Issam H. Laradji, Spandana Gella, Nicolas Gontier", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15017", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-12", + "updated_at": "2026-06-12", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15034-osguard-a-benchmark-for-safety-in-computer-use-agents.md", + "title": "\"OSGuard: A Benchmark for Safety in Computer-Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OSGuard: A Benchmark for Safety in Computer-Use Agents\"", + "authors": "Mina Mohammadmirzaei, Jeffrey Flanigan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15034", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-13", + "updated_at": "2026-06-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15079-ling-and-ring-2-6-technical-report-efficient-and-instant-agentic-intelligence-at.md", + "title": "\"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale\"", + "authors": "Ang Li, Ben Liu, Bin Han, Bin Hu, Bin Jing, Binbin Hu, Bing Li, Cai Chen, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15079", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-13", + "updated_at": "2026-06-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15152-can-agents-read-the-room-benchmarking-visual-social-intelligence-in-multimodal-s.md", + "title": "Can Agents Read the Room? Benchmarking Visual Social Intelligence in Multimodal Simulation", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can Agents Read the Room? Benchmarking Visual Social Intelligence in Multimodal Simulation", + "authors": "Shijun Wan, Xuehai Wu, Jiwen Zhang, Siyuan Wang, Zhongyu Wei", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15152", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-13", + "updated_at": "2026-06-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15242-benign-in-isolation-harmful-in-composition-security-risks-in-agent-skill-ecosyst.md", + "title": "\"Benign in Isolation, Harmful in Composition: Security Risks in Agent Skill Ecosystems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Benign in Isolation, Harmful in Composition: Security Risks in Agent Skill Ecosystems\"", + "authors": "Yi Xie, Jiawei Du, Yu Cheng, Jiuan Zhou, Zhaoxia Yin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15242", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-13", + "updated_at": "2026-06-13", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15376-coagent-concurrency-control-for-multi-agent-systems.md", + "title": "\"CoAgent: Concurrency Control for Multi-Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CoAgent: Concurrency Control for Multi-Agent Systems\"", + "authors": "Hongtao Lyu, Dingyan Zhang, Mingyu Wu, Xingda Wei, Haibo Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15376", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-13", + "updated_at": "2026-06-13", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "multi-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DC", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15591-agentic-retrieval-and-reinforcement-learned-equation-chains-a-controlled-generat.md", + "title": "\"Agentic Retrieval and Reinforcement Learned Equation Chains: A Controlled Generation Framework for Complex and Novel Physics Word Problems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic Retrieval and Reinforcement Learned Equation Chains: A Controlled Generation Framework for Complex and Novel Physics Word Problems\"", + "authors": "Tirthankar Mittra", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15591", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15609-fragfuse-bypassing-access-control-of-large-language-model-agents-via-memory-base.md", + "title": "\"FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion\"", + "authors": "Zixin Rao, Wentian Zhu, Chan Aristella Lu, Zhaorun Chen, Wei Niu, Le Guan, Bo Li, Zhen Xiang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15609", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15684-multi-agent-framework-for-time-sensitive-complementary-collaboration-in-minecraf.md", + "title": "Multi-agent Framework for Time-Sensitive Complementary Collaboration in Minecraft", + "type": "paper", + "meta": { + "type": "paper", + "title": "Multi-agent Framework for Time-Sensitive Complementary Collaboration in Minecraft", + "authors": "Juheon Yi, Jinglu Wang, Xiaoyi Zhang, Yan Lu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15684", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15709-ai-driven-framework-for-adaptive-water-network-management-with-proof-of-concept-.md", + "title": "\"AI-Driven Framework for Adaptive Water Network Management with Proof-of-Concept Implementation: Addressing Non-Revenue Water in Jordan\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AI-Driven Framework for Adaptive Water Network Management with Proof-of-Concept Implementation: Addressing Non-Revenue Water in Jordan\"", + "authors": "Mohammed Fasha, Nahel Al-Maayta, Bilal Sowan, Mohammad Athamneh, Husam Barham", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15709", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "function-calling, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15862-retailbench-benchmarking-long-horizon-reasoning-and-coherent-decision-making-of-.md", + "title": "\"RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments\"", + "authors": "Linghua Zhang, Jun Wang, Jingtong Wu, Zhisong Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15862", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15874-llm-as-code-agentic-programming-for-agent-harness.md", + "title": "\"LLM-as-Code: Agentic Programming for Agent Harness\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LLM-as-Code: Agentic Programming for Agent Harness\"", + "authors": "Junjia Qi, Zichuan Fu, Jingtong Gao, Wenlin Zhang, Hanyu Yan, Xian Wu, Xiangyu Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15874", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15903-control-plane-placement-shapes-forgetting-an-architectural-study-of-agent-memory.md", + "title": "\"Control-Plane Placement Shapes Forgetting: An Architectural Study of Agent Memory Across Thirteen System Configurations\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Control-Plane Placement Shapes Forgetting: An Architectural Study of Agent Memory Across Thirteen System Configurations\"", + "authors": "Dongxu Yang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15903", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15906-mage-rag-multigranular-adaptive-graph-evidence-for-agentic-multimodal-rag-in-lon.md", + "title": "\"MAGE-RAG: Multigranular Adaptive Graph Evidence for Agentic Multimodal RAG in Long-Document QA\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MAGE-RAG: Multigranular Adaptive Graph Evidence for Agentic Multimodal RAG in Long-Document QA\"", + "authors": "Yilong Zuo, Xunkai Li, Jing Yuan, Qiangqiang Dai, Hongchao Qin, Ronghua Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15906", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI", + "cs.CL", + "cs.DB", + "cs.MM" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15931-deeproot-a-kg-coordinated-multi-agent-system-for-therapeutic-reasoning-over-hist.md", + "title": "\"DeepRoot: A KG-Coordinated Multi-Agent System for Therapeutic Reasoning over Historical Medical Texts\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DeepRoot: A KG-Coordinated Multi-Agent System for Therapeutic Reasoning over Historical Medical Texts\"", + "authors": "Zijian Carl Ma, Sean J. Wang, Sijbren Kramer, Li Erran Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15931", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-15994-agentic-framework-for-deep-learning-workload-migration-via-in-context-learning.md", + "title": "Agentic Framework for Deep Learning workload migration via In-Context Learning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agentic Framework for Deep Learning workload migration via In-Context Learning", + "authors": "Qiyue Liang, Steven Ingram, George Vanica, Andi Gavrilescu, Newfel Harrat, Hassan Sipra, Sethuraman Sankaran", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.15994", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-14", + "updated_at": "2026-06-14", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16111-towards-pareto-optimal-tool-integrated-agents-with-pareto-ranking-policy-optimiz.md", + "title": "Towards Pareto-Optimal Tool-Integrated Agents with Pareto Ranking Policy Optimization", + "type": "paper", + "meta": { + "type": "paper", + "title": "Towards Pareto-Optimal Tool-Integrated Agents with Pareto Ranking Policy Optimization", + "authors": "Junyi Li, Xiaowei Qian, Yingyi Zhang, Wenlin Zhang, Guojing Li, Sheng Zhang, Xiao Han, Yichao Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16111", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16295-visualclaw-a-real-time-personalized-agent-for-the-physical-world.md", + "title": "\"VisualClaw: A Real-Time, Personalized Agent for the Physical World\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VisualClaw: A Real-Time, Personalized Agent for the Physical World\"", + "authors": "Haoqin Tu, Jianwen Chen, Zijun Wang, Siwei Han, Juncheng Wu, Hardy Chen, Haonian Ji, Kaiwen Xiong, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16295", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation, tool-use, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16420-transferable-self-evolving-playbooks-for-agentic-security-auditing.md", + "title": "Transferable Self-Evolving Playbooks for Agentic Security Auditing", + "type": "paper", + "meta": { + "type": "paper", + "title": "Transferable Self-Evolving Playbooks for Agentic Security Auditing", + "authors": "Ziyue Wang, Cheuk Wang Maurice Ng, Chenchen Yu, Strick Sheng, Kaihua Qin, Liyi Zhou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16420", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-safety, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16432-accord-action-conditioned-contextual-grounding-for-language-agents.md", + "title": "\"ACCORD: Action-Conditioned Contextual Grounding for Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ACCORD: Action-Conditioned Contextual Grounding for Language Agents\"", + "authors": "Lai Jiang, Cheng Qian, Zhenhailong Wang, Pan Lu, Heng Ji, Hao Peng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16432", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16481-steering-emotional-dynamics-for-art-therapy-controllable-narrative-script-genera.md", + "title": "\"Steering Emotional Dynamics for Art Therapy: Controllable Narrative Script Generation through Hierarchically Guided LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Steering Emotional Dynamics for Art Therapy: Controllable Narrative Script Generation through Hierarchically Guided LLM Agents\"", + "authors": "Suqing Wang, Qinghai Miao, Chao Guo, Yisheng Lv", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16481", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16534-generated-parallel-scalable-a-study-of-agentic-ai-generated-julia-code-on-superc.md", + "title": "Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers", + "type": "paper", + "meta": { + "type": "paper", + "title": "Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers", + "authors": "Linus Bantel, Anna-Lena Roth, Jonas Posner, Dirk Pflüger", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16534", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16576-can-llm-agents-infer-world-models-evidence-from-agentic-automata-learning.md", + "title": "Can LLM Agents Infer World Models? Evidence from Agentic Automata Learning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can LLM Agents Infer World Models? Evidence from Agentic Automata Learning", + "authors": "Reef Menaged, Gili Lior, Shauli Ravfogel, Roee Aharoni, Gabriel Stanovsky", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16576", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16591-sing-synthetic-intention-graph-for-scalable-active-tool-discovery-in-llm-agents.md", + "title": "\"SING: Synthetic Intention Graph for Scalable Active Tool Discovery in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SING: Synthetic Intention Graph for Scalable Active Tool Discovery in LLM Agents\"", + "authors": "Qiao Xiao, Haochen Shi, Yisen Gao, Wenbin Hu, Huihao Jing, Tianshi Zheng, Baixuan Xu, Ziheng Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16591", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16613-coffeebench-benchmarking-long-horizon-llm-agents-in-heterogeneous-multi-agent-ec.md", + "title": "\"CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies\"", + "authors": "Issa Sugiura, Daichi Hattori, Kazuo Araragi, Keita Ogawa, Shota Onose, Taro Makino, Teppei Usuki, Takashi Ishida", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16613", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "autonomous-agent-llm, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16659-fraudsmswalker-benchmarking-agentic-large-language-models-for-sms-to-webpage-fra.md", + "title": "\"FraudSMSWalker: Benchmarking Agentic Large Language Models for SMS-to-Webpage Fraud Detection\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"FraudSMSWalker: Benchmarking Agentic Large Language Models for SMS-to-Webpage Fraud Detection\"", + "authors": "Y. H. Zhou, Z. M. Ma, Y. J. Zhou, Y. T. Li, H. X. Xiang, Y. M. Cheng, T. L. Chen, K. J. Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16659", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16748-mypcbench-a-benchmark-for-personally-intelligent-computer-use-agents.md", + "title": "\"MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents\"", + "authors": "Lawrence Keunho Jang, Andrew Keunwoo Jang, Jing Yu Koh, Ruslan Salakhutdinov", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16748", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16774-openclaw-skill-collective-skill-tree-search-for-agentic-large-language-models.md", + "title": "\"OpenClaw-Skill: Collective Skill Tree Search for Agentic Large Language Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OpenClaw-Skill: Collective Skill Tree Search for Agentic Large Language Models\"", + "authors": "Tianyi Lin, Chuanyu Sun, Jingyi Zhang, Changxu Wei, Huanjin Yao, Shunyu Liu, Xikun Zhang, Liu Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16774", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "planning-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16802-labosbench-benchmarking-computer-use-agents-for-scientific-instrument-control.md", + "title": "\"LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control\"", + "authors": "Anqi Zou, Han Deng, Chengyu Zhang, Junquan Hu, Yu Wang, Yuxiang Xing, Aokai Zhang, Hanling Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16802", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16813-gist-cmtf-goal-state-inference-for-causal-minimal-tool-filtering-in-llm-agents.md", + "title": "\"GIST-CMTF: Goal-State Inference for Causal Minimal Tool Filtering in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GIST-CMTF: Goal-State Inference for Causal Minimal Tool Filtering in LLM Agents\"", + "authors": "Rahul Suresh Babu, Rohit Shukla", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16813", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16839-towards-llm-accelerated-rapid-reviews-for-software-tool-discovery-case-for-log-a.md", + "title": "Towards LLM Accelerated Rapid Reviews for Software Tool Discovery -- Case for Log Anomaly Detection", + "type": "paper", + "meta": { + "type": "paper", + "title": "Towards LLM Accelerated Rapid Reviews for Software Tool Discovery -- Case for Log Anomaly Detection", + "authors": "Jesse Nyyssölä, Hamza Bin Mazhar, Alexander Bakhtin, Matteo Esposito, Nana Reinikainen, Yuqing Wang, Ying Song, Davide Taibi, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16839", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-16871-human-on-the-bridge-scalable-evaluation-for-ai-agents.md", + "title": "\"Human-on-the-Bridge: Scalable Evaluation for AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Human-on-the-Bridge: Scalable Evaluation for AI Agents\"", + "authors": "Fouad Bousetouane", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.16871", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17041-benchmarking-llm-agents-on-meta-analysis-articles-from-nature-portfolio.md", + "title": "Benchmarking LLM Agents on Meta-Analysis Articles from Nature Portfolio", + "type": "paper", + "meta": { + "type": "paper", + "title": "Benchmarking LLM Agents on Meta-Analysis Articles from Nature Portfolio", + "authors": "Anzhe Xie, Weihang Su, Yujia Zhou, Yiqun Liu, Qingyao Ai", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17041", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17076-cmip-forge-an-agentic-system-that-retrieves-computes-and-self-reviews-climate-sc.md", + "title": "\"CMIP-Forge: An Agentic System that Retrieves, Computes, and Self-Reviews Climate Science\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CMIP-Forge: An Agentic System that Retrieves, Computes, and Self-Reviews Climate Science\"", + "authors": "Dmitrii Pantiukhin, Boris Shapkin, Ivan Kuznetsov, Thomas Jung, Nikolay Koldunov", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17076", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-10", + "updated_at": "2026-06-10", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "physics.ao-ph", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17114-an-evaluation-of-data-leakage-risks-in-tool-using-llm-agents-in-realistic-scenar.md", + "title": "An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios", + "type": "paper", + "meta": { + "type": "paper", + "title": "An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios", + "authors": "Hankyul Baek, Jaewon Noh, Sang Seo, Yongsu Kim, Gabriel Waikin Loh Matienzo, Young Il Kim, Ee Wei Seah, Akriti Vij", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17114", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-safety, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17246-geodisaster-benchmarking-orchestrated-agents-for-operational-disaster-geo-intell.md", + "title": "\"GeoDisaster: Benchmarking Orchestrated Agents for Operational Disaster Geo-Intelligence\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GeoDisaster: Benchmarking Orchestrated Agents for Operational Disaster Geo-Intelligence\"", + "authors": "Maram Hasan, Aman Verma, Savitra Roy, Hariseetharam Gunduboina, Daksh Jain, Muhammad Haris Khan, Subhasis Chaudhuri, Biplab Banerjee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17246", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17368-distributed-general-purpose-agent-networks-architecture-key-mechanisms-and-proto.md", + "title": "\"Distributed General-Purpose Agent Networks: Architecture, Key Mechanisms, and Prototypes\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Distributed General-Purpose Agent Networks: Architecture, Key Mechanisms, and Prototypes\"", + "authors": "Shengli Zhang, Deen Ma, Zibin Lin, Taotao Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17368", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-15", + "updated_at": "2026-06-15", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.NI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17383-model-validation-of-agentic-ai-systems-a-pomdp-based-framework-for-belief-state-.md", + "title": "\"Model Validation of Agentic AI Systems: A POMDP-Based Framework for Belief-State, Forecast, and Policy Validation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Model Validation of Agentic AI Systems: A POMDP-Based Framework for Belief-State, Forecast, and Policy Validation\"", + "authors": "Matthew Francis Dixon", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17383", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "q-fin.RM", + "cs.AI", + "cs.LG", + "stat.ML" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17449-mode-rag-manifold-outlier-diagnosis-and-energy-based-retrieval-augmented-generat.md", + "title": "\"MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation\"", + "authors": "Zehang Wei, Jiaxin Dai, Jiamin Yan, Xiang Xiang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17449", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.CV", + "cs.LG", + "cs.MM" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17453-mapsatisfybench-benchmarking-satisfaction-aware-map-agents-through-behavior-grou.md", + "title": "\"MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors\"", + "authors": "Lubin Bai, Mengyu Cao, Sixue Wang, Zhongwei Wan, Yue Pan, Jiale Hou, Xiang Li, Xiuyuan Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17453", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17459-can-llms-be-ceos-benchmarking-strategic-resource-reallocation-with-multi-role-ag.md", + "title": "Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation", + "authors": "Yuyang Dai, Xueqing Peng, Lingfei Qian, Zhuohan Xie", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17459", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "agent-evaluation, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17573-cordon-semantic-transactions-for-tool-using-llm-agents.md", + "title": "\"Cordon: Semantic Transactions for Tool-Using LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Cordon: Semantic Transactions for Tool-Using LLM Agents\"", + "authors": "Zheng Chen, Hanqing Liu, Duling Xu, Dong Dong, Jialin Li, Bangzheng Pu, Jidong Zhai", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17573", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.OS", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-17680-envrl-learn-from-environment-dynamics-in-agentic-reinforcement-learning.md", + "title": "\"EnvRL: Learn from Environment Dynamics in Agentic Reinforcement Learning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EnvRL: Learn from Environment Dynamics in Agentic Reinforcement Learning\"", + "authors": "Zhitong Wang, Songze Li, Hao Peng, Shuzheng Si, Yi Wang, Maosong Sun, Juanzi Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.17680", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18023-loopcoder-v2-only-loop-once-for-efficient-test-time-computation-scaling.md", + "title": "\"LoopCoder-v2: Only Loop Once for Efficient Test-Time Computation Scaling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LoopCoder-v2: Only Loop Once for Efficient Test-Time Computation Scaling\"", + "authors": "Jian Yang, Shawn Guo, Wei Zhang, Tianyu Zheng, Yaxin Du, Haau-Sing Li, Jiajun Wu, Yue Song, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18023", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18037-provenanceguard-source-aware-factuality-verification-for-mcp-based-llm-agents.md", + "title": "\"ProvenanceGuard: Source-Aware Factuality Verification for MCP-Based LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ProvenanceGuard: Source-Aware Factuality Verification for MCP-Based LLM Agents\"", + "authors": "Ander Alvarez, Santhiya Rajan, Samuel Mugel, Román Orús", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18037", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18051-compositional-skill-routing-for-llm-agents-decompose-retrieve-and-compose.md", + "title": "\"Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose\"", + "authors": "Xueping Gao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18051", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18068-agentic-ai-based-framework-for-mitigating-premature-diagnostic-handoff-and-silen.md", + "title": "Agentic AI-based Framework for Mitigating Premature Diagnostic Handoff and Silent Hallucination in Healthcare Applications", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agentic AI-based Framework for Mitigating Premature Diagnostic Handoff and Silent Hallucination in Healthcare Applications", + "authors": "Divyansh Srivastava, Shreya Ghosh, Anshul Verma, Rajkumar Buyya", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18068", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18142-your-ai-travel-agent-would-book-you-a-bullfight-an-agentic-benchmark-for-implici.md", + "title": "\"Your AI Travel Agent Would Book You a Bullfight: An Agentic Benchmark for Implicit Animal Welfare in Frontier AI Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Your AI Travel Agent Would Book You a Bullfight: An Agentic Benchmark for Implicit Animal Welfare in Frontier AI Models\"", + "authors": "Jasmine Brazilek, Joel Christoph, Maheep Chaudhary, Oliver Tullio, Carol Kline, Miles Tidmarsh, Arturs Kanepajs", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18142", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.CY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18272-mitigating-anchoring-bias-in-llm-based-agents-for-energy-efficient-6g-autonomous.md", + "title": "Mitigating Anchoring Bias in LLM-Based Agents for Energy-Efficient 6G Autonomous Networks", + "type": "paper", + "meta": { + "type": "paper", + "title": "Mitigating Anchoring Bias in LLM-Based Agents for Energy-Efficient 6G Autonomous Networks", + "authors": "Hatim Chergui, Claudia Carballo González, Farhad Rezazadeh, Merouane Debbah", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18272", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-05", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.NI", + "cs.AI", + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18356-safeclawbench-separating-semantic-audit-evidence-and-sandbox-harm-in-tool-using-.md", + "title": "\"SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents\"", + "authors": "Yuchuan Tian, Mengyu Zheng, Haocheng Mei, Ye Yuan, Chao Xu, Xinghao Chen, Hanting Chen, Yu Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18356", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-safety, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18363-guava-an-effective-and-universal-harness-for-embodied-manipulation.md", + "title": "\"Guava: An Effective and Universal Harness for Embodied Manipulation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Guava: An Effective and Universal Harness for Embodied Manipulation\"", + "authors": "Haowen Liu, Xirui Li, Shaoxiong Yao, Peng Shi, Tianyi Zhou, Jia-Bin Huang, Furong Huang, Jiayuan Mao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18363", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "embodied-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agentic-ai, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18406-coremem-riemannian-retrieval-and-fisher-guided-distillation-for-long-term-memory.md", + "title": "\"CoreMem: Riemannian Retrieval and Fisher-Guided Distillation for Long-Term Memory in Dialogue Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CoreMem: Riemannian Retrieval and Fisher-Guided Distillation for Long-Term Memory in Dialogue Agents\"", + "authors": "Jiaqi Chen, Yongqin Zeng, Shaoshen Chen, Yijian Zhang, Hai-Tao Zheng, Chunxia Ma, XiuTeng Zhou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18406", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18467-toolchain-crc-conformal-risk-control-for-agentic-ai-under-retrieval-and-tool-use.md", + "title": "\"ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift\"", + "authors": "Jeffery Opoku, David Banahene", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18467", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "stat.ML", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation, agentic-ai, ai-agent, rag-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18502-towards-scalable-customization-and-deployment-of-multi-agent-systems-for-enterpr.md", + "title": "Towards Scalable Customization and Deployment of Multi-Agent Systems for Enterprise Applications", + "type": "paper", + "meta": { + "type": "paper", + "title": "Towards Scalable Customization and Deployment of Multi-Agent Systems for Enterprise Applications", + "authors": "Paresh Dashore, Shreyas Kulkarni, Uttam Gurram, Nadia Bathaee, Kartik Balasubramaniam, Genta Indra Winata, Sambit Sahu, Shi-Xiong Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18502", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18619-code-augur-agentic-vulnerability-detection-via-specification-inference.md", + "title": "\"Code-Augur: Agentic Vulnerability Detection via Specification Inference\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Code-Augur: Agentic Vulnerability Detection via Specification Inference\"", + "authors": "Zhengxiong Luo, Mehtab Zafar, Dylan Wolff, Abhik Roychoudhury", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18619", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18671-hansel-extracting-breadcrumbs-from-web-agent-trajectories-for-interactive-verifi.md", + "title": "\"HANSEL: Extracting Breadcrumbs from Web Agent Trajectories for Interactive Verification\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HANSEL: Extracting Breadcrumbs from Web Agent Trajectories for Interactive Verification\"", + "authors": "Yujin Zhang, Daye Nam", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18671", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "planning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "ai-agent, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18789-poweragentbench-ss-a-benchmark-for-agentic-ai-in-power-system-steady-state-studi.md", + "title": "\"PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies\"", + "authors": "Costas Mylonas, Magda Foti, Andrea Pomarico, Matheus Duarte, Qian Zhang, Emmanouel Varvarigos", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18789", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "24", + "collection_queries": "agentic-ai, planning-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18829-gatemem-benchmarking-memory-governance-in-multi-principal-shared-memory-agents.md", + "title": "\"GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents\"", + "authors": "Zhe Ren, Yibo Yang, Yimeng Chen, Zijun Zhao, Benshuo Fu, Zhihao Shu, Bingjie Zhang, Yangyang Xu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18829", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-18950-rtsgamebench-an-rts-benchmark-for-strategic-reasoning-by-vision-language-models.md", + "title": "\"RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models\"", + "authors": "San Kim, Daechul Ahn, Reokyoung Kim, Hyeonbeom Choi, Seungyeon Jwa, Jonghyun Choi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.18950", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19063-pypiline-malicious-pypi-package-detection-via-suspicious-api-knowledge-and-agent.md", + "title": "\"PYPILINE: Malicious PyPI Package Detection via Suspicious API Knowledge and Agent Workflow\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PYPILINE: Malicious PyPI Package Detection via Suspicious API Knowledge and Agent Workflow\"", + "authors": "Siyuan Pang, Yepeng Yao, Zhengwei Jiang, Zijing Fan, Haozhe Li, Baoxu Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19063", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19242-runtime-compliance-verification-for-ai-agents.md", + "title": "Runtime Compliance Verification for AI Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Runtime Compliance Verification for AI Agents", + "authors": "Nafiseh Kahani, Masoud Barati, Diana Addae", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19242", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent, function-calling, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19245-txbench-pp-analyzing-ai-agent-performance-on-small-molecule-preclinical-pharmaco.md", + "title": "\"TxBench-PP: Analyzing AI Agent Performance on Small-Molecule Preclinical Pharmacology\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TxBench-PP: Analyzing AI Agent Performance on Small-Molecule Preclinical Pharmacology\"", + "authors": "Hannah Le, Ramesh Ramasamy, Alex Urrutia, Mahsa Yazdani, Tim Proctor, Kenny Workman", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19245", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19409-openrath-session-centered-runtime-state-for-agent-systems.md", + "title": "\"OpenRath: Session-Centered Runtime State for Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OpenRath: Session-Centered Runtime State for Agent Systems\"", + "authors": "Fukang Wen, Zhijie Wang, Ruilin Xu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19409", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.PL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19464-deontic-policies-for-runtime-governance-of-agentic-ai-systems.md", + "title": "Deontic Policies for Runtime Governance of Agentic AI Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Deontic Policies for Runtime Governance of Agentic AI Systems", + "authors": "Anupam Joshi, Tim Finin, Karuna Pande Joshi, Lalana Kagal", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19464", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai, autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19613-staminabench-stress-testing-coding-agents-over-100-interaction-turns.md", + "title": "\"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns\"", + "authors": "Vlad Sobal, Shuo Yang, Yuting Zhang, Wei Xia, Stefano Soatto", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19613", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19704-beyond-static-leaderboards-predictive-validity-for-the-evaluation-of-llm-agents.md", + "title": "\"Beyond Static Leaderboards: Predictive Validity for the Evaluation of LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Static Leaderboards: Predictive Validity for the Evaluation of LLM Agents\"", + "authors": "Dhaval C. Patel, Kaoutar El Maghraoui, Shuxin Lin, Yusheng Li, Tianjun Feng, Chun-Yi Tsai, Yihan Sun, Wei Alexander Xin, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19704", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19787-oragentbench-can-llm-agents-solve-challenging-operations-research-tasks-end-to-e.md", + "title": "\"ORAgentBench: Can LLM Agents Solve Challenging Operations Research Tasks End to End?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ORAgentBench: Can LLM Agents Solve Challenging Operations Research Tasks End to End?\"", + "authors": "Jiajun Li, Mingshu Cai, Yixuan Li, Yu Ding, Ran Hou, Guanyu Nie, Xiongwei Han, Wanyuan Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19787", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19812-human-on-the-loop-orchestration-for-ai-assisted-legal-discovery.md", + "title": "Human-on-the-Loop Orchestration for AI-Assisted Legal Discovery", + "type": "paper", + "meta": { + "type": "paper", + "title": "Human-on-the-Loop Orchestration for AI-Assisted Legal Discovery", + "authors": "Anushree Sinha, Srivaths Ranganathan, Abhishek Dharmaratnakar, Debanshu Das", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19812", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19852-prompt-plan-extract-zero-shot-agentic-llms-workflows-for-lung-pathology-extracti.md", + "title": "\"Prompt, Plan, Extract: Zero-Shot Agentic LLMs Workflows for Lung Pathology Extraction from Clinical Narratives\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Prompt, Plan, Extract: Zero-Shot Agentic LLMs Workflows for Lung Pathology Extraction from Clinical Narratives\"", + "authors": "Aman Pathak, Cheng Peng, Mengxian Lyu, Ziyi Chen, Reema Solan, Sankalp Talankar, Yasir Khan, Hiren Mehta, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19852", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19899-measuring-biological-capabilities-and-risks-of-ai-agents.md", + "title": "Measuring Biological Capabilities and Risks of AI Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Measuring Biological Capabilities and Risks of AI Agents", + "authors": "Patricia Paskov, Jeffrey Lee, Kyle Brady, Alyssa Worland", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19899", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-evaluation, agentic-ai, ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19926-memgui-agent-an-end-to-end-long-horizon-mobile-gui-agent-with-proactive-context-.md", + "title": "\"MemGUI-Agent: An End-to-End Long-Horizon Mobile GUI Agent with Proactive Context Management\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemGUI-Agent: An End-to-End Long-Horizon Mobile GUI Agent with Proactive Context Management\"", + "authors": "Guangyi Liu, Gao Wu, Congxiao Liu, Pengxiang Zhao, Liang Liu, Mading Li, Qi Zhang, Mengyan Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19926", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19930-mobileforge-annotation-free-adaptation-for-mobile-gui-agents-with-hierarchical-f.md", + "title": "\"MobileForge: Annotation-Free Adaptation for Mobile GUI Agents with Hierarchical Feedback-Guided Policy Optimization\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MobileForge: Annotation-Free Adaptation for Mobile GUI Agents with Hierarchical Feedback-Guided Policy Optimization\"", + "authors": "Guangyi Liu, Pengxiang Zhao, Gao Wu, Yiwen Yin, Mading Li, Liang Liu, Congxiao Liu, Zhang Qi, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19930", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-19980-enpire-agentic-robot-policy-self-improvement-in-the-real-world.md", + "title": "\"ENPIRE: Agentic Robot Policy Self-Improvement in the Real World\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ENPIRE: Agentic Robot Policy Self-Improvement in the Real World\"", + "authors": "\"Wenli Xiao, Jia Xie, Tonghe Zhang, Haotian Lin, Letian \\\"Max\\\" Fu, Haoru Xue, Jalen Lu, Yi Yang, et al.\"", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.19980", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "coding-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20023-when-lower-privileges-suffice-investigating-over-privileged-tool-selection-in-ll.md", + "title": "\"When Lower Privileges Suffice: Investigating Over-Privileged Tool Selection in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When Lower Privileges Suffice: Investigating Over-Privileged Tool Selection in LLM Agents\"", + "authors": "Kaiyue Yang, Yuyan Bu, Jingwei Yi, Yuchi Wang, Biyu Zhou, Juntao Dai, Songlin Hu, Yaodong Yang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20023", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20041-ai-economist-agent-an-agentic-framework-for-model-grounded-economic-analysis-wit.md", + "title": "\"AI Economist Agent: An Agentic Framework for Model-Grounded Economic Analysis with RAG, Knowledge Graphs, and Large Language Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AI Economist Agent: An Agentic Framework for Model-Grounded Economic Analysis with RAG, Knowledge Graphs, and Large Language Models\"", + "authors": "Masahiro Kato", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20041", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "econ.GN", + "cs.AI", + "cs.LG", + "q-fin.GN" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "ai-agent, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20047-pacms-submodular-context-selection-as-a-pluggable-engine-for-llm-agents.md", + "title": "\"PACMS: Submodular Context Selection as a Pluggable Engine for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PACMS: Submodular Context Selection as a Pluggable Engine for LLM Agents\"", + "authors": "Manu Ghulyani, Arunabh Singh, Karan Bharadwaj, Ankit Nath, Suranjan Goswami", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20047", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20243-phoenix-safe-github-issue-resolution-via-multi-agent-llms.md", + "title": "\"Phoenix: Safe GitHub Issue Resolution via Multi-Agent LLMs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Phoenix: Safe GitHub Issue Resolution via Multi-Agent LLMs\"", + "authors": "Kipngeno Koech, Muhammad Adam, Baimam Boukar Jean Jacques, Joao Barros", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20243", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "multi-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20401-poweragentbench-dyn-a-benchmark-for-agentic-ai-in-power-system-dynamic-studies.md", + "title": "\"PowerAgentBench-Dyn: A Benchmark for Agentic AI in Power System Dynamic Studies\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PowerAgentBench-Dyn: A Benchmark for Agentic AI in Power System Dynamic Studies\"", + "authors": "Qian Zhang, Andrea Pomarico, Costas Mylonas, Magda Foti, Alberto Berizzi, Le Xie", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20401", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "24", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20470-analyzing-defensive-misdirection-against-model-guided-automated-attacks-on-agent.md", + "title": "Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems", + "authors": "Reza Soosahabi, Vivek Namsani", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20470", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20479-groundcontrol-anticipating-navigation-failures-in-vision-language-agents-via-tra.md", + "title": "\"GroundControl: Anticipating Navigation Failures in Vision-Language Agents via Trajectory-Consistent Uncertainty Estimates\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GroundControl: Anticipating Navigation Failures in Vision-Language Agents via Trajectory-Consistent Uncertainty Estimates\"", + "authors": "Nastaran Darabi, Divake Kumar, Sina Tayebati, Devashri Naik, Amit Ranjan Trivedi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20479", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20510-efficient-and-sound-probabilistic-verification-for-ai-agents.md", + "title": "Efficient and Sound Probabilistic Verification for AI Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Efficient and Sound Probabilistic Verification for AI Agents", + "authors": "Alaia Solko-Breslin, Pramod Kaushik Mudrakarta, Mihai Christodorescu, Somesh Jha, Krishnamurthy Dj Dvijotham", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20510", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20512-probe-and-refine-tuning-of-repository-guidance-for-coding-agents.md", + "title": "Probe-and-Refine Tuning of Repository Guidance for Coding Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Probe-and-Refine Tuning of Repository Guidance for Coding Agents", + "authors": "Asa Shepard, Jeannie Albrecht", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20512", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "coding-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20515-s-agent-spatial-tool-use-elicits-reasoning-for-spatial-intelligence.md", + "title": "\"S-Agent: Spatial Tool-Use Elicits Reasoning for Spatial Intelligence\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"S-Agent: Spatial Tool-Use Elicits Reasoning for Spatial Intelligence\"", + "authors": "Yalun Dai, Hao Li, Shulin Tian, Runmao Yao, Yuhao Dong, Fangzhou Hong, Zhaoxi Chen, Fangfu Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20515", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20573-aona-a-comprehensive-architecture-and-workflow-design-for-global-agentic-collabo.md", + "title": "\"AONA: A Comprehensive Architecture and Workflow Design for Global Agentic Collaboration\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AONA: A Comprehensive Architecture and Workflow Design for Global Agentic Collaboration\"", + "authors": "Jinliang Xu, Runkai Zhu, Bingqi Li, Fanjie Nie, Jin Li, Jiagui Xie", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20573", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-04-30", + "updated_at": "2026-04-30", + "status": "queued", + "relevance": "high", + "topics": [ + "multi-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.NI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20629-specialize-roles-mix-deployments-pushing-the-cost-accuracy-frontier-of-llm-agent.md", + "title": "\"Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams\"", + "authors": "Yinsicheng Jiang, Liang Cheng, Yeqi Huang, Yufan Zhao, Zhan Lu, Li Dong, Wenda Li, Edoardo Ponti, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20629", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-05-28", + "updated_at": "2026-05-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20717-mirage-stealthy-visual-prompt-injection-for-vulnerability-detection-in-web-agent.md", + "title": "\"MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents\"", + "authors": "Xuelong Dai, Jianyu Ma, Boyang Ma, Biwei Yan, Yijun Yang, Yue Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20717", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-16", + "updated_at": "2026-06-16", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20785-fara-1-5-scalable-learning-environments-for-computer-use-agents.md", + "title": "\"Fara-1.5: Scalable Learning Environments for Computer Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Fara-1.5: Scalable Learning Environments for Computer Use Agents\"", + "authors": "Ahmed Awadallah, Sahil Gupta, Yash Lara, Yadong Lu, Hussein Mozannar, Akshay Nambi, Zach Nussbaum, Yash Pandya, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20785", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20922-think-twice-before-you-act-protecting-llm-agents-against-tool-description-poison.md", + "title": "\"Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning\"", + "authors": "Shanghao Shi, Xiao Wang, Chaoyu Zhang, Hao Li, Wenjing Lou, Thomas Hou, Yevgeniy Vorobeychik, Chongjie Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20922", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20950-power-systems-agent-benchmark-executable-evaluation-of-ai-agents-in-electric-pow.md", + "title": "\"Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering\"", + "authors": "Sergei Trashchenkov", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20950", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation, ai-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-20954-learning-what-not-to-forget-long-horizon-agent-memory-from-a-few-kilobytes-of-le.md", + "title": "\"Learning What Not to Forget: Long-Horizon Agent Memory from a Few Kilobytes of Learning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Learning What Not to Forget: Long-Horizon Agent Memory from a Few Kilobytes of Learning\"", + "authors": "Nusrat Jahan Lia, Aritra Mazumder", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.20954", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-18", + "updated_at": "2026-06-18", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21013-agentic-time-machine-as-an-infrastructure-for-future-event-forecasting.md", + "title": "Agentic Time Machine as an Infrastructure for Future-Event Forecasting", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agentic Time Machine as an Infrastructure for Future-Event Forecasting", + "authors": "Jingyi Chai, Bingyang Zheng, Xiangrui Liu, Hao Lu, Zihang Zhou, Tianchen Wang, Kemeng Zhang, Siheng Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21013", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21123-a-multi-agent-audit-framework-for-high-stakes-reasoning-evaluation-and-interpret.md", + "title": "\"A Multi-Agent Audit Framework for High-Stakes Reasoning: Evaluation and Interpretability in Clinical Mental Health Screening\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"A Multi-Agent Audit Framework for High-Stakes Reasoning: Evaluation and Interpretability in Clinical Mental Health Screening\"", + "authors": "Jingchen Ye, Yanpei Yu, Luyao Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21123", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21129-agenticos-an-intent-oriented-secure-operating-system-architecture-for-autonomous.md", + "title": "\"AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents\"", + "authors": "Zhen Zhao, Yu Zhang, Yanpeng Zhu, Jia Wang, Songqiao Tao, Xin Cheng, Jiexin Gao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21129", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.OS" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "ai-agent, autonomous-agent-llm, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21228-sakana-fugu-technical-report.md", + "title": "Sakana Fugu Technical Report", + "type": "paper", + "meta": { + "type": "paper", + "title": "Sakana Fugu Technical Report", + "authors": "Yujin Tang, Edoardo Cetin, Jinglue Xu, Qi Sun, Stefan Nielsen, Vincent Richard, Haruto Goda, Iaroslav Tymchenko, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21228", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "multi-agent", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21401-swarmx-agentic-scheduling-for-low-latency-agentic-systems.md", + "title": "\"SwarmX: Agentic Scheduling for Low-Latency Agentic Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SwarmX: Agentic Scheduling for Low-Latency Agentic Systems\"", + "authors": "Yeqi Huang, Yanwei Ye, Guomin Chen, Wenhao Su, Bin Gong, Jialian Li, Zhan Lu, Yangshen Deng, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21401", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DC", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21409-don-t-blindly-trust-it-how-unreliable-feedback-breaks-tool-using-llm-agents.md", + "title": "\"Don't Blindly Trust It: How Unreliable Feedback Breaks Tool-Using LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Don't Blindly Trust It: How Unreliable Feedback Breaks Tool-Using LLM Agents\"", + "authors": "Chubin Zhang, Zhenglin Wan, Xingrui Yu, Pengfei Zhou, Wangbo Zhao, Jingxuan Wu, Yaxin Zhou, Ivor Tsang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21409", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21445-autoras-learning-robust-agentic-systems-with-primitive-representations.md", + "title": "\"AutoRAS: Learning Robust Agentic Systems with Primitive Representations\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AutoRAS: Learning Robust Agentic Systems with Primitive Representations\"", + "authors": "Yang Yue, Xuancheng Zhu, Yuyang Ma, Guoshun Nan, Zihan Dou, Jingru Shan, Congyu Guo, Ji Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21445", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21553-dissecting-agentic-rag-a-component-ablation-for-multi-hop-qa-with-a-local-7b-mod.md", + "title": "\"Dissecting Agentic RAG: A Component Ablation for Multi-Hop QA with a Local 7B Model\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Dissecting Agentic RAG: A Component Ablation for Multi-Hop QA with a Local 7B Model\"", + "authors": "Sheroz Shaikh", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21553", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21565-composing-verifiable-conceptual-models-via-building-blocks-towards-design-time-v.md", + "title": "\"Composing Verifiable Conceptual Models via Building Blocks: Towards Design-Time Verification of Agentic AI Workflows\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Composing Verifiable Conceptual Models via Building Blocks: Towards Design-Time Verification of Agentic AI Workflows\"", + "authors": "Noe Y. Flandre, Alexander C. Nwala, Philippe J. Giabbanelli", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21565", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21627-counsel-a-meta-evaluation-dataset-for-agentic-tasks.md", + "title": "\"Counsel: A Meta-Evaluation Dataset for Agentic Tasks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Counsel: A Meta-Evaluation Dataset for Agentic Tasks\"", + "authors": "Sashank Pisupati, Henry Broomfield, Eujeong Choi, Antonia Calvi, Charlie Wang, Roman Engeler, Max Bartolo, Patrick Lewis", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21627", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-evaluation, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21649-evoembedding-evolvable-representations-for-long-context-retrieval-and-agentic-me.md", + "title": "\"EvoEmbedding: Evolvable Representations for Long-Context Retrieval and Agentic Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EvoEmbedding: Evolvable Representations for Long-Context Retrieval and Agentic Memory\"", + "authors": "Chang Nie, Chaoyou Fu, Junlan Feng, Caifeng Shan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21649", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory, agentic-ai, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21710-privacyalign-contextual-privacy-alignment-for-llm-agents.md", + "title": "\"PrivacyAlign: Contextual Privacy Alignment for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PrivacyAlign: Contextual Privacy Alignment for LLM Agents\"", + "authors": "Manveer Singh Tamber, Abhay Puri, Marc-Etienne Brunet, Perouz Taslakian, Jimmy Lin, Spandana Gella", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21710", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21732-safe-to-check-unsafe-to-use-relinking-at-the-compression-boundary-of-llm-agents.md", + "title": "\"Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents\"", + "authors": "Zesen Liu, Zihan Zhang, Dongdong She", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21732", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21740-training-the-orchestrator-a-supervised-approach-to-end-to-end-pddl-planning-with.md", + "title": "\"Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents\"", + "authors": "Rajesh Mangannavar, Zachary Coalson, Pranay Dugar, Prasad Tadepalli", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21740", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-19", + "updated_at": "2026-06-19", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21836-agentdse-reasoning-augmented-architectural-design-space-exploration.md", + "title": "\"AgentDSE: Reasoning-Augmented Architectural Design Space Exploration\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentDSE: Reasoning-Augmented Architectural Design Space Exploration\"", + "authors": "Chenyu Wang, Jiahe Caroline Shi, David Kong, Duane S. Boning, Zishen Wan, Yilun Du, Vijay Janapa Reddi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21836", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-20", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21842-agent-assisted-side-channel-attacks-on-non-prefix-kv-cache-in-rag.md", + "title": "Agent-Assisted Side-Channel Attacks on Non-Prefix KV Cache in RAG", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agent-Assisted Side-Channel Attacks on Non-Prefix KV Cache in RAG", + "authors": "He Sun, Shinan Liu, Siyuan Ma, Junhao Li, Mingjun Xiao, Wenhao Jiang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21842", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-20", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21877-agentriskbom-a-risk-scoping-security-bill-of-materials-for-agentic-ai-systems.md", + "title": "\"AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems\"", + "authors": "Srimonti Dutta, Akshata Kishore Moharir", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21877", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-20", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CR", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "agentic-ai, ai-agent, rag-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-21963-holmes-multimodal-agentic-diagnosis-for-mixed-language-mobile-crashes-at-industr.md", + "title": "\"Holmes: Multimodal Agentic Diagnosis for Mixed-Language Mobile Crashes at Industrial Scale\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Holmes: Multimodal Agentic Diagnosis for Mixed-Language Mobile Crashes at Industrial Scale\"", + "authors": "Jia Li, Wenyuan Ma, Ting Peng, Haibin Zheng, Yuetang Deng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.21963", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-20", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22030-nous-a-predictive-world-model-for-long-term-agent-memory.md", + "title": "\"Nous: A Predictive World Model for Long-Term Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Nous: A Predictive World Model for Long-Term Agent Memory\"", + "authors": "Pranav Singh", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22030", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-20", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.IR", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22082-codeteam-an-llm-powered-multi-agent-framework-for-repository-level-code-generati.md", + "title": "\"CodeTeam: An LLM-Powered Multi-Agent Framework for Repository-Level Code Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CodeTeam: An LLM-Powered Multi-Agent Framework for Repository-Level Code Generation\"", + "authors": "Yifei Wang, Ruiyin Li, Peng Liang, Qiong Feng, Zengyang Li, Mojtaba Shahin, Arif Ali Khan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22082", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-20", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22110-traceview-interactive-visualization-of-agentic-program-repair-trajectories.md", + "title": "\"TraceView: Interactive Visualization of Agentic Program Repair Trajectories\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TraceView: Interactive Visualization of Agentic Program Repair Trajectories\"", + "authors": "Amirali Sajadi, Tu Nguyen, Kimmie Huynh, Esteban Parra, Preetha Chatterjee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22110", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-20", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22151-novelty-aware-agentic-retrieval-comparing-research-contributions-through-structu.md", + "title": "\"Novelty-Aware Agentic Retrieval: Comparing Research Contributions Through Structured Multi-Step Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Novelty-Aware Agentic Retrieval: Comparing Research Contributions Through Structured Multi-Step Reasoning\"", + "authors": "Shou-Tzu Han", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22151", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-20", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22263-revelio-cost-efficient-agentic-memory-safety-vulnerability-detection-for-reposit.md", + "title": "\"Revelio: Cost-Efficient Agentic Memory Safety Vulnerability Detection For Repository-Scale Codebases\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Revelio: Cost-Efficient Agentic Memory Safety Vulnerability Detection For Repository-Scale Codebases\"", + "authors": "Yiwei Hou, Hao Wang, Muxi Lyu, Marius Momeu, Eric Nguyen, Taige Yang, Koushik Sen, Dawn Song, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22263", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-20", + "updated_at": "2026-06-20", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.MA", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-memory, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22330-hypothesis-driven-skill-optimization-for-llm-agents.md", + "title": "Hypothesis-Driven Skill Optimization for LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Hypothesis-Driven Skill Optimization for LLM Agents", + "authors": "Fangxin Shang, Yehui Yang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22330", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-06-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22388-planbench-xl-evaluating-long-horizon-planning-of-llm-tool-use-agents-in-large-sc.md", + "title": "\"PlanBench-XL: Evaluating Long-Horizon Planning of LLM Tool-Use Agents in Large-Scale Tool Ecosystems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PlanBench-XL: Evaluating Long-Horizon Planning of LLM Tool-Use Agents in Large-Scale Tool Ecosystems\"", + "authors": "Jiayu Liu, Qihan Lin, Cheng Qian, Rui Wang, Emre Can Acikgoz, Xiaocheng Yang, Jiateng Liu, Zhenhailong Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22388", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-06-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22417-code-isn-t-memory-a-structural-codebase-index-inside-a-coding-agent.md", + "title": "\"Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent\"", + "authors": "Ishaan Bhola, Adithyan Krishnan, Sravanth Kurmala, Mukunda NS", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22417", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-06-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22484-governed-ai-assisted-engineering-graduated-human-oversight-for-agentic-code-gene.md", + "title": "\"Governed AI-Assisted Engineering: Graduated Human Oversight for Agentic Code Generation in Regulated Domains\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Governed AI-Assisted Engineering: Graduated Human Oversight for Agentic Code Generation in Regulated Domains\"", + "authors": "Richard Kang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22484", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22495-grounded-scaling-why-agentic-ai-needs-deterministic-environments.md", + "title": "\"Grounded Scaling: Why Agentic AI Needs Deterministic Environments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Grounded Scaling: Why Agentic AI Needs Deterministic Environments\"", + "authors": "Liang Ding, Xintong Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22495", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-06-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "embodied-agent", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22557-macagentbench-benchmarking-ai-agents-on-real-world-macos-desktop.md", + "title": "\"MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop\"", + "authors": "Yikun Fu, Bowen Fu, Zhenyu Wu, Shuang Cheng, Xiaowei Sun, Bowen Yang, Zehao Li, Yibo Zhao, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22557", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-06-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "agent-evaluation, ai-agent, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22610-paperclaw-harnessing-agents-for-autonomous-research-and-human-in-the-loop-refine.md", + "title": "\"PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement\"", + "authors": "Weiwei Ye, Hangchen Liu, Dongyuan Li, Renhe Jiang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22610", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-06-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22647-raven-agentic-rag-for-automated-vulnerability-repair.md", + "title": "\"RAVEN: Agentic RAG for Automated Vulnerability Repair\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RAVEN: Agentic RAG for Automated Vulnerability Repair\"", + "authors": "Varun Gadey, Zijie Liu, Alexandra Dmitrienko", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22647", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-06-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.LG", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22673-agentlens-interpretable-safety-steering-via-mechanistic-subspaces-for-multi-turn.md", + "title": "\"AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent\"", + "authors": "Weidi Luo, Qiming Zhang, Yihao Quan, Mingyu Jin, Jie Cai, Chaowei Xiao, Jingcheng Niu, Zhen Xiang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22673", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-06-21", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-safety, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22678-rigorbench-benchmarking-engineering-process-discipline-in-autonomous-ai-coding-a.md", + "title": "\"RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents\"", + "authors": "Meher Bhaskar Madiraju, Meher Sai Preetam Madiraju", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22678", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-21", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22737-groundeval-a-deterministic-replacement-for-llm-as-judge-in-stateful-agent-evalua.md", + "title": "\"GroundEval: A Deterministic Replacement for LLM-as-Judge in Stateful Agent Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GroundEval: A Deterministic Replacement for LLM-as-Judge in Stateful Agent Evaluation\"", + "authors": "Jeffrey Flynt", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22737", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22741-grade-graph-representation-of-llm-agent-dependency-and-execution.md", + "title": "\"GRADE: Graph Representation of LLM Agent Dependency and Execution\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GRADE: Graph Representation of LLM Agent Dependency and Execution\"", + "authors": "Yue Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22741", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22844-ramem-contextual-reinstatement-for-long-term-agentic-memory.md", + "title": "\"RaMem: Contextual Reinstatement for Long-term Agentic Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RaMem: Contextual Reinstatement for Long-term Agentic Memory\"", + "authors": "Wei Yang, Bryce Kan, Shixuan Li, Li Li, Yuehan Qin, Jiate Li, Paul Bogdan, Jesse Thomason", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22844", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22864-when-auc-0-998-is-not-enough-a-candidate-evaluation-protocol-for-hidden-state-pr.md", + "title": "\"When AUC 0.998 Is Not Enough: A Candidate Evaluation Protocol for Hidden-State Probes of Indirect Prompt Injection in Multimodal Computer-Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When AUC 0.998 Is Not Enough: A Candidate Evaluation Protocol for Hidden-State Probes of Indirect Prompt Injection in Multimodal Computer-Use Agents\"", + "authors": "Yanhang Li, Zhichao Fan, Zexin Zhuang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22864", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22948-envs-environment-native-verified-search-for-long-horizon-gui-agents.md", + "title": "\"ENVS: Environment-Native Verified Search for Long-Horizon GUI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ENVS: Environment-Native Verified Search for Long-Horizon GUI Agents\"", + "authors": "Yincheng Zhou, Athena Zhuoming Zhong, Shijie Zhang, Kevin Zhang, Teresa Xiaotao Shang, Shanghang Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22948", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-22953-plans-don-t-persist-why-context-management-is-load-bearing-for-llm-agents.md", + "title": "\"Plans Don't Persist: Why Context Management Is Load Bearing for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Plans Don't Persist: Why Context Management Is Load Bearing for LLM Agents\"", + "authors": "Aman Mehta, Anupam Datta", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.22953", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23032-ipo-finance-agent-benchmark-of-llm-financial-analysts-beyond-finance-agent-v2-wi.md", + "title": "\"IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO\"", + "authors": "Mostapha Benhenda", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23032", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "q-fin.GN" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23130-understanding-the-in-security-of-vibe-coded-applications.md", + "title": "Understanding the (In)Security of Vibe-Coded Applications", + "type": "paper", + "meta": { + "type": "paper", + "title": "Understanding the (In)Security of Vibe-Coded Applications", + "authors": "Junquan Deng, Zhiyu Fan, Ruijie Meng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23130", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23195-memory-contagion-cross-temporal-propagation-of-evaluator-bias-via-agent-memory.md", + "title": "\"Memory Contagion: Cross-Temporal Propagation of Evaluator Bias via Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Memory Contagion: Cross-Temporal Propagation of Evaluator Bias via Agent Memory\"", + "authors": "Zewen Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23195", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23283-towards-root-memories-benchmarking-and-enhancing-implicit-logical-memory-retriev.md", + "title": "\"Towards Root Memories: Benchmarking and Enhancing Implicit Logical Memory Retrieval for Personalized LLMs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Towards Root Memories: Benchmarking and Enhancing Implicit Logical Memory Retrieval for Personalized LLMs\"", + "authors": "Hongxun Ding, Xiang Yu, Chengbing Wang, Jianfei Xiao, Keqin Bao, Wenjie Wang, Xiangnan He", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23283", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23327-videoagent-all-in-one-framework-for-video-understanding-and-editing.md", + "title": "\"VideoAgent: All-in-One Framework for Video Understanding and Editing\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VideoAgent: All-in-One Framework for Video Understanding and Editing\"", + "authors": "Hengji Zhou, Lingxuan Huang, Jian Wang, Bing Zhou, Si Wu, Lianghao Xia, Chao Huang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23327", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23343-group-selection-promotes-prosocial-prompts-in-populations-of-llm-agents.md", + "title": "Group Selection Promotes Prosocial Prompts in Populations of LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Group Selection Promotes Prosocial Prompts in Populations of LLM Agents", + "authors": "Luis Celiktemel, Edward Eichhorn, Levin Brinkmann, Robin Schimmelpfennig, Aron Vallinder, Yaomin Jiang, Edward Hughes, Iyad Rahwan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23343", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "computer-use", + "multi-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23565-holoagent-0-a-unified-embodied-agent-framework-with-3d-spatial-memory.md", + "title": "\"HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory\"", + "authors": "Xiaolin Zhou, Liu Liu, Tingyang Xiao, Wei Feng, Fa Fu, Xinrui Meng, Xinjie Wang, Jialiang Han, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23565", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO", + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23664-mas-promptbench-when-does-prompt-optimization-improve-multi-agent-llm-systems.md", + "title": "\"MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems?\"", + "authors": "Juyang Bai, Laixi Shi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23664", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23752-esaa-conversational-an-event-sourced-memory-layer-for-continuity-handoff-and-cur.md", + "title": "\"ESAA-Conversational: An Event-Sourced Memory Layer for Continuity, Handoff, and Curation Across Heterogeneous LLM Coding Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ESAA-Conversational: An Event-Sourced Memory Layer for Continuity, Handoff, and Curation Across Heterogeneous LLM Coding Agents\"", + "authors": "Elzo Brito dos Santos Filho", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23752", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23764-emergent-relational-order-in-llm-agent-societies-from-collective-affect-to-autho.md", + "title": "\"Emergent Relational Order in LLM Agent Societies: From Collective Affect to Authority Stratification\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Emergent Relational Order in LLM Agent Societies: From Collective Affect to Authority Stratification\"", + "authors": "Zhiyuan Ji, Xinyu Chen, Ziqi Dai, Shiyun Tang, Chunyu Wei, Yueguo Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23764", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23927-rift-bench-dynamic-red-teaming-for-agentic-ai-systems.md", + "title": "\"RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems\"", + "authors": "Yarin Yerushalmi Levi, Roy Betser, Amit Giloni, Lidor Erez, Itay Gershon, Oren Rachmil, Sindhu Padakandla, Roman Vainshtein", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23927", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-23991-critique-of-agent-model.md", + "title": "Critique of Agent Model", + "type": "paper", + "meta": { + "type": "paper", + "title": "Critique of Agent Model", + "authors": "Eric Xing, Mingkai Deng, Jinyu Hou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.23991", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG", + "cs.MA", + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agentic-ai, ai-agent, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24193-skychain-intelligence-a-blockchain-secured-multi-agent-drl-framework-for-low-alt.md", + "title": "\"SkyChain Intelligence: A Blockchain-Secured Multi-Agent DRL Framework for Low-Altitude Embodied Artificial Intelligence\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SkyChain Intelligence: A Blockchain-Secured Multi-Agent DRL Framework for Low-Altitude Embodied Artificial Intelligence\"", + "authors": "Haoxiang Luo, Tianqi Jiang, Ruichen Zhang, Yinqiu Liu, Gang Sun, Hongfang Yu, Abbas Jamalipour, Dong In Kim", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24193", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "embodied-agent", + "multi-agent", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.NI", + "cs.DC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24235-sp-mind-an-autonomous-reasoning-agent-for-spatial-proteomics-analysis.md", + "title": "\"SP-Mind: An Autonomous Reasoning Agent for Spatial Proteomics Analysis\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SP-Mind: An Autonomous Reasoning Agent for Spatial Proteomics Analysis\"", + "authors": "Yucheng Yuan, Yuanfeng Ji, Zhongxiao Li, Ruijiang Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24235", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24322-securing-llm-agent-long-term-memory-against-poisoning-non-malleable-origin-bound.md", + "title": "\"Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees\"", + "authors": "Yedidel Louck", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24322", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24402-poisoned-playbooks-demystifying-knowledge-poisoning-effects-on-ai-security-agent.md", + "title": "\"Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents\"", + "authors": "Juho Park, Hyunmin Choi, Kevin Nam", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24402", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "ai-agent, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24416-agentic-ai-for-bilevel-long-term-optimization-of-policy-driven-physical-layer-sy.md", + "title": "Agentic AI for Bilevel Long-Term Optimization of Policy-Driven Physical Layer Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agentic AI for Bilevel Long-Term Optimization of Policy-Driven Physical Layer Systems", + "authors": "Bingnan Xiao, Chenhao Yang, Wei Ni, Xin Wang, Tony Q. S. Quek", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24416", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24437-rem-moa-reasoning-memory-sustains-mixture-of-agents-scaling.md", + "title": "\"ReM-MoA: Reasoning Memory Sustains Mixture-of-Agents Scaling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ReM-MoA: Reasoning Memory Sustains Mixture-of-Agents Scaling\"", + "authors": "Heng Ping, Arijit Bhattacharjee, Peiyu Zhang, Shixuan Li, Wei Yang, Ali Jannesari, Nesreen Ahmed, Paul Bogdan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24437", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24453-bayesian-control-for-coding-agents.md", + "title": "Bayesian control for coding agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Bayesian control for coding agents", + "authors": "Theodore Papamarkou, Vladislav Smirnov, Viktor Mazanov, Artem Vazhentsev, Preslav Nakov, Timothy Baldwin, Artem Shelmanov", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24453", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "coding-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24515-reinforcement-learning-for-computer-use-agents-with-autonomous-evaluation.md", + "title": "Reinforcement Learning for Computer-Use Agents with Autonomous Evaluation", + "type": "paper", + "meta": { + "type": "paper", + "title": "Reinforcement Learning for Computer-Use Agents with Autonomous Evaluation", + "authors": "Marta Sumyk, Oleksandr Kosovan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24515", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24525-viscritic-visual-state-comparison-as-process-reward-for-gui-agents.md", + "title": "\"VisCritic: Visual State Comparison as Process Reward for GUI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VisCritic: Visual State Comparison as Process Reward for GUI Agents\"", + "authors": "Jiachen Qian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24525", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24535-governed-shared-memory-for-multi-agent-llm-systems.md", + "title": "Governed Shared Memory for Multi-Agent LLM Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Governed Shared Memory for Multi-Agent LLM Systems", + "authors": "Yanki Margalit, Nurit Cohen-Inger, Erni Avram, Ran Taig, Oded Margalit", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24535", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-memory, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24551-gui-vs-cli-execution-bottlenecks-in-screen-only-and-skill-mediated-computer-use-.md", + "title": "\"GUI vs. CLI: Execution Bottlenecks in Screen-Only and Skill-Mediated Computer-Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GUI vs. CLI: Execution Bottlenecks in Screen-Only and Skill-Mediated Computer-Use Agents\"", + "authors": "Xiao Zhou, Siyue Zhang, Yilun Zhao, Jinbiao Wei, Tingyu Song, Arman Cohan, Chen Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24551", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24595-memprobe-probing-long-term-agent-memory-via-hidden-user-state-recovery.md", + "title": "\"MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery\"", + "authors": "Enze Ma, Yufan Zhou, Wei-Chieh Huang, Jie Yang, Huanhuan Ma, Zixuan Wang, Chengze Li, Chunyu Miao, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24595", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24597-qwen-agentworld-language-world-models-for-general-agents.md", + "title": "\"Qwen-AgentWorld: Language World Models for General Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Qwen-AgentWorld: Language World Models for General Agents\"", + "authors": "Yuxin Zuo, Zikai Xiao, Li Sheng, Fei Huang, Jianhong Tu, Yuxuan Liu, Tianyi Tang, Xiaomeng Hu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24597", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24623-privacy-preserving-rag-via-multi-agent-semantic-rewriting-achieving-confidential.md", + "title": "\"Privacy-Preserving RAG via Multi-Agent Semantic Rewriting: Achieving Confidentiality Without Compromising Contextual Fidelity\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Privacy-Preserving RAG via Multi-Agent Semantic Rewriting: Achieving Confidentiality Without Compromising Contextual Fidelity\"", + "authors": "Yuanhe Zhao, Tianyu Zhang, Huafei Xing, Derek F. Wong, Jianbin Li, Tao Fang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24623", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24626-safari-scaling-long-horizon-agentic-fault-attribution-via-active-investigation.md", + "title": "\"SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation\"", + "authors": "Chenyang Zhu, Jiayu Yao, Kushal Chawla, Youbing Yin, Nathan Wolfe, Pengshan Cai, Jingyu Wu, Spencer Hong, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24626", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "autonomous-agent-llm, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24649-agentic-collaborative-cognition-for-zero-shot-3d-understanding.md", + "title": "Agentic Collaborative Cognition for Zero-Shot 3D Understanding", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agentic Collaborative Cognition for Zero-Shot 3D Understanding", + "authors": "Wenxin Wang, Bo Zhang, Feng Chen, Zixuan Wang, Wen Li, Changsheng Li, Yinjie Lei", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24649", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24689-automated-summarization-of-software-documents-an-llm-based-multi-agent-approach.md", + "title": "\"Automated Summarization of Software Documents: An LLM-based Multi-Agent Approach\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Automated Summarization of Software Documents: An LLM-based Multi-Agent Approach\"", + "authors": "Duc S. H. Nguyen, Minh T. Nguyen, Phuong T. Nguyen, Juri Di Rocco, Davide Di Ruscio", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24689", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24694-supplynet-supporting-visual-exploratory-learning-in-supply-chain-via-contextual-.md", + "title": "\"SupplyNet: Supporting Visual Exploratory Learning in Supply Chain via Contextual Multi-Agent Simulation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SupplyNet: Supporting Visual Exploratory Learning in Supply Chain via Contextual Multi-Agent Simulation\"", + "authors": "Yanjia Li, Kelcy Kexin Han, Tianrui Hu, Yi-Fan Cao, Huamin Qu, Sicheng Song", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24694", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "multi-agent", + "rag", + "reasoning", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24775-are-we-ready-for-an-agent-native-memory-system.md", + "title": "Are We Ready For An Agent-Native Memory System?", + "type": "paper", + "meta": { + "type": "paper", + "title": "Are We Ready For An Agent-Native Memory System?", + "authors": "Wei Zhou, Xuanhe Zhou, Shaokun Han, Hongming Xu, Guoliang Li, Zhiyu Li, Feiyu Xiong, Fan Wu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24775", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.DB", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24779-deepbd-a-grounded-agentic-workflow-for-variant-prioritization-and-diagnosis-of-g.md", + "title": "\"DeepBD: A Grounded Agentic Workflow for Variant Prioritization and Diagnosis of Genetic Birth Defects\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DeepBD: A Grounded Agentic Workflow for Variant Prioritization and Diagnosis of Genetic Birth Defects\"", + "authors": "Shiyu Li, Ziqi Yan, Zhihao Wu, Jielong Lu, Weiran Liao, Jiajun Yu, Genjie Li, Zeyu Chu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24779", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "q-bio.GN", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24820-sherloc-structured-diagnostic-localization-for-code-repair-agents.md", + "title": "\"SHERLOC: Structured Diagnostic Localization for Code Repair Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SHERLOC: Structured Diagnostic Localization for Code Repair Agents\"", + "authors": "Hovhannes Tamoyan, Sean Narenthiran, Erik Arakelyan, Mira Mezini, Boris Ginsburg", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24820", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "coding-agent, multi-agent-llm, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24839-grading-the-grader-lessons-from-evaluating-an-agentic-data-analysis-system.md", + "title": "\"Grading the Grader: Lessons from Evaluating an Agentic Data Analysis System\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Grading the Grader: Lessons from Evaluating an Agentic Data Analysis System\"", + "authors": "Tian Zheng, Kai-Tai Hsu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24839", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "stat.AP" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24855-openthoughts-agent-data-recipes-for-agentic-models.md", + "title": "\"OpenThoughts-Agent: Data Recipes for Agentic Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OpenThoughts-Agent: Data Recipes for Agentic Models\"", + "authors": "Negin Raoof, Richard Zhuang, Marianna Nezhurina, Etash Guha, Atula Tejaswi, Ryan Marten, Charlie F. Ruan, Tyler Griggs, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24855", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24937-the-hitchhiker-s-guide-to-agentic-ai-from-foundations-to-systems.md", + "title": "\"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems\"", + "authors": "Haggai Roitman", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24937", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-22", + "updated_at": "2026-06-22", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.IR", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "25", + "collection_queries": "agentic-ai, multi-agent-llm, rag-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-24976-diagnosing-and-mitigating-compounding-failures-in-agentic-persuasion-via-taxonom.md", + "title": "Diagnosing and Mitigating Compounding Failures in Agentic Persuasion via Taxonomic Strategy Retrieval", + "type": "paper", + "meta": { + "type": "paper", + "title": "Diagnosing and Mitigating Compounding Failures in Agentic Persuasion via Taxonomic Strategy Retrieval", + "authors": "Sana Ayromlou, Purvi Sehgal, Pradyumna Narayana", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.24976", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25115-forget-to-improve-on-device-llm-agent-continual-learning-via-budget-curated-memo.md", + "title": "\"Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory\"", + "authors": "Beining Wu, Zihao Ding, Jun Huang, Yanxiao Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25115", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.NI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25139-buildrix-an-open-platform-for-sharing-and-benchmarking-agentic-ai-skills-in-buil.md", + "title": "\"Buildrix: An Open Platform for Sharing and Benchmarking Agentic AI Skills in Building Engineering\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Buildrix: An Open Platform for Sharing and Benchmarking Agentic AI Skills in Building Engineering\"", + "authors": "Zixin Jiang, Bing Dong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25139", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25161-trustmem-learning-trustworthy-memory-consolidation-for-llm-agents-with-long-term.md", + "title": "\"TRUSTMEM: Learning Trustworthy Memory Consolidation for LLM Agents with Long-Term Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TRUSTMEM: Learning Trustworthy Memory Consolidation for LLM Agents with Long-Term Memory\"", + "authors": "Tianyu Yang, Sudipta Paul, Vijay Srinivasan, Vivek Kulkarni, Srinivas Chappidi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25161", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25189-actplane-programmable-os-level-policy-enforcement-for-agent-harnesses.md", + "title": "\"ActPlane: Programmable OS-Level Policy Enforcement for Agent Harnesses\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ActPlane: Programmable OS-Level Policy Enforcement for Agent Harnesses\"", + "authors": "Yusheng Zheng, Tianyuan Wu, Quanzhi Fu, Tong Yu, Wenan Mao, Tao Ma, Dan Williams, Wei Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25189", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.OS" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25191-to-isolate-or-to-score-model-adaptive-assessment-for-cost-efficient-multi-agent-.md", + "title": "To Isolate or to Score? Model-Adaptive Assessment for Cost-Efficient Multi-Agent RAG", + "type": "paper", + "meta": { + "type": "paper", + "title": "To Isolate or to Score? Model-Adaptive Assessment for Cost-Efficient Multi-Agent RAG", + "authors": "Jungseob Lee, Chanjun Park, Heuiseok Lim", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25191", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25195-sok-ai-secure-code-generation-progress-pitfalls-and-paths-forward.md", + "title": "\"SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward\"", + "authors": "Rupam Patir, Keyan Guo, Haipeng Cai, Hongxin Hu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25195", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agentic-ai, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25206-raven-long-horizon-reasoning-navigation-with-a-visuo-spatio-temporal-memory.md", + "title": "\"RAVEN: Long-Horizon Reasoning & Navigation with a Visuo-Spatio-Temporal Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RAVEN: Long-Horizon Reasoning & Navigation with a Visuo-Spatio-Temporal Memory\"", + "authors": "Yixun Hu, Zhicheng Zheng, Lihan Zha, Chunwei Xing, Rajdeep Singh, Omar Hossain, Antonio Loquercio, Dhruv Shah", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25206", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-23", + "updated_at": "2026-06-23", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25334-bridging-the-post-discharge-gap-a-traceable-multi-agent-framework-for-safe-and-c.md", + "title": "\"Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care\"", + "authors": "Runwei Guan, Yi Zhou, Heyi Lin, Jinjing Zhu, Mingyuan Hou, Yang Yang, Fang Yuan, Xiaohong Lin, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25334", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25358-agentic-knowledge-tracing-a-multi-agent-llm-architecture-for-stealth-assessment-.md", + "title": "\"Agentic Knowledge Tracing: A Multi-Agent LLM Architecture for Stealth Assessment of Financial Literacy in Serious Games\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic Knowledge Tracing: A Multi-Agent LLM Architecture for Stealth Assessment of Financial Literacy in Serious Games\"", + "authors": "Gabriel Santos, Rita Julia, Marcelo Nascimento", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25358", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25361-memory-makes-the-difference-evaluating-how-different-memory-roles-shape-conversa.md", + "title": "\"Memory Makes the Difference: Evaluating How Different Memory Roles Shape Conversational Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Memory Makes the Difference: Evaluating How Different Memory Roles Shape Conversational Agents\"", + "authors": "Yuxin Wang, Paul Thomas, Zhiwei Yu, Yuan Gao, Saeed Hassanpour, Soroush Vosoughi, Robert Sim, Nick Craswell", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25361", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25400-brainagent-a-large-language-model-driven-multi-agent-framework-for-autonomous-br.md", + "title": "\"BrainAgent: A Large Language Model-Driven Multi-Agent Framework for Autonomous Brain Signal Understanding\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"BrainAgent: A Large Language Model-Driven Multi-Agent Framework for Autonomous Brain Signal Understanding\"", + "authors": "Yangxuan Zhou, Sha Zhao, Jiquan Wang, Shijian Li, Gang Pan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25400", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25484-from-causal-discovery-to-implementation-an-agentic-ai-framework-for-e-scooter-mo.md", + "title": "\"From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities\"", + "authors": "Meng Jin, Melanie Handrich, Simone Martinenz, Nicholas Hoeser, Ziyue Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25484", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY", + "econ.GN", + "stat.AP" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25514-unlocking-model-potentials-through-adaptive-multi-agent-scaffolding-for-efficien.md", + "title": "Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution", + "type": "paper", + "meta": { + "type": "paper", + "title": "Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution", + "authors": "Yang Chen, Aliya Ahmad, Yiheng Zhou, Reyhaneh Jabbarvand", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25514", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25519-quantization-inflates-reasoning-token-inflation-as-a-hidden-cost-of-low-bit-reas.md", + "title": "\"Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models\"", + "authors": "Xinyu Lian, Walid Krichene, Beichen Huang, Masahiro Tanaka, Olatunji Ruwase, Li Zhang, Minjia Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25519", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25588-intenttester-intent-driven-multi-agent-framework-for-cross-library-test-migratio.md", + "title": "\"IntentTester: Intent-Driven Multi-agent Framework for Cross-Library Test Migration\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"IntentTester: Intent-Driven Multi-agent Framework for Cross-Library Test Migration\"", + "authors": "Yi Gao, Ziyuan Zhang, Xing Hu, Xiaohu Yang, Xin Xia", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25588", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25622-probabilistic-agents-in-deterministic-audits-evaluating-multi-agent-systems-for-.md", + "title": "\"Probabilistic Agents in Deterministic Audits: Evaluating Multi-Agent Systems for Automated Audits Based on the German IT-Grundschutz\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Probabilistic Agents in Deterministic Audits: Evaluating Multi-Agent Systems for Automated Audits Based on the German IT-Grundschutz\"", + "authors": "Lea Roxanne Muth, Marian Margraf", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25622", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25651-medguards-multi-agent-system-for-reliable-medical-error-detection-and-correction.md", + "title": "\"MedGuards: Multi-Agent System for Reliable Medical Error Detection and Correction\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MedGuards: Multi-Agent System for Reliable Medical Error Detection and Correction\"", + "authors": "Congbo Ma, Hu Wang, Yichun Zhang, Farah E. Shamout", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25651", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25656-is-graphrag-needed-from-basic-rag-to-graph-agentic-solutions-with-context-optimi.md", + "title": "Is GraphRAG Needed? From Basic RAG to Graph-/Agentic Solutions with Context Optimization", + "type": "paper", + "meta": { + "type": "paper", + "title": "Is GraphRAG Needed? From Basic RAG to Graph-/Agentic Solutions with Context Optimization", + "authors": "Long Chen, Ryan Razkenari, Yuxuan Zhou, Yuan Tian, Rahul Ghosh, Venkatesh Pappakrishnan, Disha Ahuja, Vidya Sagar Ravipati", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25656", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25705-gui-agent-guided-exploration-of-user-sensitive-screens.md", + "title": "\"GUI agent: Guided Exploration of User-Sensitive Screens\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GUI agent: Guided Exploration of User-Sensitive Screens\"", + "authors": "Aradhana Nayak, Mussadiq Nazeer, Wang Peng, Feng Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25705", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25760-uncertainty-quantification-for-computer-use-agents-a-benchmark-across-vision-lan.md", + "title": "\"Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets\"", + "authors": "Divake Kumar, Sina Tayebati, Devashri Naik, Amanda Sofie Rios, Nilesh Ahuja, Omesh Tickoo, Ranganath Krishnan, Amit Ranjan Trivedi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25760", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL", + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25819-beyond-function-calling-benchmarking-tool-using-agents-under-tool-environment-un.md", + "title": "\"Beyond Function Calling: Benchmarking Tool-Using Agents under Tool-Environment Unreliability\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Function Calling: Benchmarking Tool-Using Agents under Tool-Environment Unreliability\"", + "authors": "Yang Tian, Zhengpeng Shi, Yu Zhou, Bo Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25819", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-25899-manipulation-is-task-dependent-a-multi-axis-multi-environment-evaluation-of-fron.md", + "title": "\"Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs\"", + "authors": "Adeeb Zaman, Erik Nordby, Fred Heiding", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.25899", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26057-the-unfireable-safety-kernel-execution-time-ai-alignment-for-ai-agents-and-other.md", + "title": "\"The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems\"", + "authors": "Seth Dobrin, Łukasz Chmiel", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26057", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CR", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26203-agentic-analysis-for-agentic-infrastructure-an-llm-powered-pipeline-for-comparat.md", + "title": "\"Agentic Analysis for Agentic Infrastructure: An LLM-Powered Pipeline for Comparative Governance of DAO and Corporate AI Protocols\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic Analysis for Agentic Infrastructure: An LLM-Powered Pipeline for Comparative Governance of DAO and Corporate AI Protocols\"", + "authors": "Yutian Wang, Luyao Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26203", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CY", + "cs.MA", + "cs.SI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai, ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26205-knowledge-augmented-agentic-ai-for-mental-health-medication-information-seeking.md", + "title": "Knowledge-augmented Agentic AI for Mental Health Medication Information Seeking", + "type": "paper", + "meta": { + "type": "paper", + "title": "Knowledge-augmented Agentic AI for Mental Health Medication Information Seeking", + "authors": "Huizi Yu, Jian Liu, Wenkong Wang, Lingyao Li, Jiayan Zhou, Zhaoqian Xue, Xiang Li, Xinxin Lin, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26205", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26216-cyberchainbench-can-ai-agents-secure-smart-contracts-against-real-world-on-chain.md", + "title": "\"CyberChainBench: Can AI Agents Secure Smart Contracts Against Real-World On-Chain Vulnerabilities?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CyberChainBench: Can AI Agents Secure Smart Contracts Against Real-World On-Chain Vulnerabilities?\"", + "authors": "Jintao Huang, Fengqing Jiang, Radha Poovendran, Zhiqiang Lin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26216", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety, ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26300-the-verification-horizon-no-silver-bullet-for-coding-agent-rewards.md", + "title": "\"The Verification Horizon: No Silver Bullet for Coding Agent Rewards\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Verification Horizon: No Silver Bullet for Coding Agent Rewards\"", + "authors": "Binghai Wang, Chenlong Zhang, Dayiheng Liu, Jiajun Zhang, Jiawei Chen, Mingze Li, Mouxiang Chen, Rongyao Fang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26300", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26346-how-do-tool-augmented-llm-agents-perform-on-real-world-energy-analytics-tasks.md", + "title": "How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?", + "type": "paper", + "meta": { + "type": "paper", + "title": "How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?", + "authors": "David Akinpelu, Akintonde Abbas, Rereloluwa Alimi, Ayodeji Lana", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26346", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26356-instruction-bleed-cross-module-interference-in-prompt-composed-agentic-systems.md", + "title": "\"Instruction Bleed: Cross-Module Interference in Prompt-Composed Agentic Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Instruction Bleed: Cross-Module Interference in Prompt-Composed Agentic Systems\"", + "authors": "Ching-Yu Lin, Yifan Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26356", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.IR", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation, agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26403-profilefoundry-a-synthetic-person-object-substrate-for-privacy-memory-and-tool-u.md", + "title": "\"ProfileFoundry: A Synthetic Person-Object Substrate for Privacy, Memory, and Tool-Use Evaluation in LLM Agent\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ProfileFoundry: A Synthetic Person-Object Substrate for Privacy, Memory, and Tool-Use Evaluation in LLM Agent\"", + "authors": "Sriram Selvam, Anneswa Ghosh", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26403", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26453-optimizing-cuda-like-a-human-micro-profiling-tools-as-expert-surrogates-for-llm-.md", + "title": "\"Optimizing CUDA like a Human: Micro-Profiling Tools as Expert Surrogates for LLM-Based GPU Kernel Optimization\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Optimizing CUDA like a Human: Micro-Profiling Tools as Expert Surrogates for LLM-Based GPU Kernel Optimization\"", + "authors": "Jiading Gai, Shuai Zhang, Kaj Bostrom, Jin Huang, Vihang Patil, Haoyang Fang, Bernie Wang, Huzefa Rangwala, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26453", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26479-adaptive-evaluation-of-out-of-band-defenses-against-prompt-injection-in-llm-agen.md", + "title": "Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents", + "authors": "Praneeth Narisetty, Shiva Nagendra Babu Kore, Uday Kumar Reddy Kattamanchi, Jayaram Kumarapu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26479", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26511-temporal-validity-in-retrieval-memory-eliminating-stale-fact-errors-for-ai-agent.md", + "title": "\"Temporal Validity in Retrieval Memory: Eliminating Stale-Fact Errors for AI Agents over Evolving Knowledge\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Temporal Validity in Retrieval Memory: Eliminating Stale-Fact Errors for AI Agents over Evolving Knowledge\"", + "authors": "Neeraj Yadav", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26511", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.ET", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "ai-agent, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26524-vigil-runtime-enforcement-of-behavioral-specifications-in-ai-agent-skills.md", + "title": "\"VIGIL: Runtime Enforcement of Behavioral Specifications in AI Agent Skills\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VIGIL: Runtime Enforcement of Behavioral Specifications in AI Agent Skills\"", + "authors": "Ying Li, Yanju Chen, Hongbo Wen, Bosi Zhang, Hanzhi Liu, Peiran Wang, Yu Feng, Yuan Tian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26524", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26614-hilsva-design-and-evaluation-of-a-human-in-the-loop-agentic-system-for-scientifi.md", + "title": "\"HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization\"", + "authors": "Kuangshi Ai, Patrick Phuoc Do, Chaoli Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26614", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.HC", + "cs.AI", + "cs.GR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "23", + "collection_queries": "multi-agent-llm, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26627-agents-that-know-too-much-a-data-centric-survey-of-privacy-in-llm-agents.md", + "title": "\"Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents\"", + "authors": "Nada Lahjouji, Ashwin Gerard Colaco", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26627", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26649-autoformalization-of-agent-instructions-into-policy-as-code.md", + "title": "Autoformalization of Agent Instructions into Policy-as-Code", + "type": "paper", + "meta": { + "type": "paper", + "title": "Autoformalization of Agent Instructions into Policy-as-Code", + "authors": "Adam Mondl, Matthew Maisel, John H. Brock", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26649", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26721-knowledge-based-pull-requests-a-trusted-workflow-for-agent-mediated-knowledge-co.md", + "title": "\"Knowledge-Based Pull Requests: A Trusted Workflow for Agent-Mediated Knowledge Collaboration\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Knowledge-Based Pull Requests: A Trusted Workflow for Agent-Mediated Knowledge Collaboration\"", + "authors": "Xinyu Zhang, Weiwei Sun", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26721", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26758-egg-an-expert-guided-agent-framework-for-kernel-generation.md", + "title": "\"EGG: An Expert-Guided Agent Framework for Kernel Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EGG: An Expert-Guided Agent Framework for Kernel Generation\"", + "authors": "Yaochen Han, Ke Fan, Hongxu Jiang, Wanqi Xu, Weiyu Xie, Runhua Zhang, Chenhui Zhu, Yixiang Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26758", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "memory", + "multi-agent", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26793-mirror-novelty-constrained-memory-guided-mcts-red-teaming-for-agentic-rag.md", + "title": "\"MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG\"", + "authors": "Inderjeet Singh, Andrés Murillo, Motoyoshi Sekiya, Yuki Unno, Junichi Suga", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26793", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26806-memory-depth-not-memory-access-selective-parametric-consolidation-for-long-runni.md", + "title": "\"Memory Depth, Not Memory Access: Selective Parametric Consolidation for Long-Running Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Memory Depth, Not Memory Access: Selective Parametric Consolidation for Long-Running Language Agents\"", + "authors": "Haoliang Han", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26806", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26883-econsimulacra-a-digital-twin-platform-of-socio-economic-systems-powered-by-llm-a.md", + "title": "\"EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents\"", + "authors": "Ryuji Hashimoto, Masahiro Kaneko, Kentaro Ueda, Takehiro Takayanagi, Kiyoshi Izumi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26883", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26918-diagnosing-task-insensitivity-in-language-agents.md", + "title": "Diagnosing Task Insensitivity in Language Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Diagnosing Task Insensitivity in Language Agents", + "authors": "Jingyu Liu, Xiaopeng Wu, Kehan Chen, Chuan Yu, Yong Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26918", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26924-a-deterministic-control-plane-for-llm-coding-agents.md", + "title": "A Deterministic Control Plane for LLM Coding Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Deterministic Control Plane for LLM Coding Agents", + "authors": "Padmaraj Madatha", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26924", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-26960-toward-agentic-sysadmin-rethinking-system-administration-with-ai-agents.md", + "title": "\"Toward Agentic SysAdmin: Rethinking System Administration with AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Toward Agentic SysAdmin: Rethinking System Administration with AI Agents\"", + "authors": "Gianmaria Frigo, Davide Saladino, Alberto Castagnaro, Francesco Marchiori, Denis Donadel, Luca Pajola, Mauro Conti", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.26960", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.NI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27009-semantic-early-stopping-for-iterative-llm-agent-loops.md", + "title": "Semantic Early-Stopping for Iterative LLM Agent Loops", + "type": "paper", + "meta": { + "type": "paper", + "title": "Semantic Early-Stopping for Iterative LLM Agent Loops", + "authors": "Sahil Shrivastava", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27009", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27154-openrca-2-0-from-outcome-labels-to-causal-process-supervision.md", + "title": "\"OpenRCA 2.0: From Outcome Labels to Causal Process Supervision\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OpenRCA 2.0: From Outcome Labels to Causal Process Supervision\"", + "authors": "\"Aoyang Fang, Yifan Yang, Jin'ao Shang, Qisheng Lu, Junjielung Xu, Rui Wang, Songhan Zhang, Yuzhong Zhang, et al.\"", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27154", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27243-nova-a-verification-aware-agent-harness-for-architecture-evolution-in-industrial.md", + "title": "\"NOVA: A Verification-Aware Agent Harness for Architecture Evolution in Industrial Recommender Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"NOVA: A Verification-Aware Agent Harness for Architecture Evolution in Industrial Recommender Systems\"", + "authors": "Shaohua Liu, Liang Fang, Yilong Sun, Shudong Huang, Qingsong Luo, Shaoxin Liu, Xiaoyang Chen, Dongqiang Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27243", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "memory", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27330-empowering-gui-agents-via-autonomous-experience-exploration-and-hindsight-experi.md", + "title": "Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning", + "authors": "Tianyi Men, Zhuoran Jin, Pengfei Cao, Yubo Chen, Kang Liu, Jun Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27330", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.CV", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27350-chia-an-open-source-framework-for-principled-agentic-ai-driven-hardware-software.md", + "title": "\"CHIA: An open-source framework for principled, agentic AI-driven hardware/software co-design research\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CHIA: An open-source framework for principled, agentic AI-driven hardware/software co-design research\"", + "authors": "Angela Cui, Ferran Hermida-Rivera, Jack Toubes, Raghav Gupta, Jim Fang, Chengyi Lux Zhang, Ella Schwarz, Junha Kim, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27350", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27397-sidconarena-an-environment-evaluating-agents-in-open-ended-positive-sum-bargaini.md", + "title": "\"SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game\"", + "authors": "Yeqi Feng, Yuxin Chen, Tianxing He", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27397", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-24", + "updated_at": "2026-06-24", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.AI", + "cs.GT" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27406-towards-evaluation-of-implicit-software-world-models-in-coding-llms.md", + "title": "Towards Evaluation of Implicit Software World Models in Coding LLMs", + "type": "paper", + "meta": { + "type": "paper", + "title": "Towards Evaluation of Implicit Software World Models in Coding LLMs", + "authors": "Egor Bogomolov, Yaroslav Zharov", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27406", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "reasoning", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "ai-agent, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27416-glite-arf-verifier-driven-research-with-parallel-llm-coding-agents.md", + "title": "\"Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents\"", + "authors": "Vassili Philippov, Pavel Katunin, Dmitry Andreev, Igor Ostanin, Anton Nikolaev", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27416", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27472-supersede-diagnosing-and-training-the-memory-update-gap-in-llm-agents.md", + "title": "\"Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents\"", + "authors": "Vedant Patel", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27472", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27483-internalizing-the-future-a-unified-agentic-training-paradigm-for-world-model-pla.md", + "title": "\"Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning\"", + "authors": "Xuan Zhang, Zhijian Zhou, Lingfeng Qiao, Yulei Qin, Ke Li, Xing Sun, Xiaoyu Tan, Chao Qu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27483", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27492-queenbee-planner-skill-evolving-communication-topologies-for-token-efficient-llm.md", + "title": "\"QueenBee Planner: Skill-Evolving Communication Topologies for Token-Efficient LLM Multi-Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"QueenBee Planner: Skill-Evolving Communication Topologies for Token-Efficient LLM Multi-Agent Systems\"", + "authors": "Congjia Tian, Yuhang Yao, Jiaming Cui", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27492", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27499-dmv-bench-diagnosing-long-horizon-multimodal-agents-visual-memory-with-incidenta.md", + "title": "\"DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection\"", + "authors": "Yujin Tang, Chenming Shang, Ruize Xu, Nikhil Singh", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27499", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27632-yuvion-llm-an-adversarially-aware-large-language-model-for-content-and-ai-safety.md", + "title": "\"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety\"", + "authors": "Ting Ma, Xiufeng Huang, Benlei Cui, Xiaowen Xu, Shikai Qiu, Ruijie Jian, Hongxing Li, Guanghui Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27632", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27806-agent-vs-parametric-world-models-hybrid-planning-for-reliable-language-agents.md", + "title": "\"Agent vs. Parametric World Models: Hybrid Planning for Reliable Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agent vs. Parametric World Models: Hybrid Planning for Reliable Language Agents\"", + "authors": "Xinyuan Song, Zekun Cai", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27806", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27929-when-multi-robot-systems-meet-agentic-ai-towards-embodied-collective-intelligenc.md", + "title": "\"When Multi-Robot Systems Meet Agentic AI:Towards Embodied Collective Intelligence\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When Multi-Robot Systems Meet Agentic AI:Towards Embodied Collective Intelligence\"", + "authors": "Yuxuan Yan, Yuanyuan Jia, Qianqian Yang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27929", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-27990-advancedshellm-a-stateful-multi-agent-llm-honeypot-for-ssh-deception.md", + "title": "\"AdvancedShelLM: A Stateful Multi-Agent LLM Honeypot for SSH Deception\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AdvancedShelLM: A Stateful Multi-Agent LLM Honeypot for SSH Deception\"", + "authors": "Muris Sladić, Eman Alibalić, Veronica Valeros, Carlos Catania, Sebastian Garcia", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.27990", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28011-from-detection-to-action-using-llm-agents-for-fault-tolerant-control.md", + "title": "\"From Detection to Action: Using LLM Agents for Fault-Tolerant Control\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Detection to Action: Using LLM Agents for Fault-Tolerant Control\"", + "authors": "Javal Vyas, Milapji Singh Gill, Artan Markaj, Felix Gehlhoff, Mehmet Mercangöz", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28011", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "24", + "collection_queries": "agentic-ai, llm-agent, multi-agent-llm, planning-agent, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28061-toolprivacybench-benchmarking-purpose-bound-privacy-in-tool-using-llm-agents.md", + "title": "\"ToolPrivacyBench: Benchmarking Purpose-Bound Privacy in Tool-Using LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ToolPrivacyBench: Benchmarking Purpose-Bound Privacy in Tool-Using LLM Agents\"", + "authors": "Shijing Hu, Liang Liu, Zhu Meng, Zhicheng Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28061", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "agent-evaluation, function-calling, llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28182-llawco-learning-laws-of-cooperation-for-modeling-embodied-multi-agent-behavior.md", + "title": "\"LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior\"", + "authors": "Qinhong Zhou, Chuang Gan, Anoop Cherian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28182", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CV", + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28187-gbc-gradient-based-connections-for-optimizing-multi-agent-systems.md", + "title": "\"GBC: Gradient-Based Connections for Optimizing Multi-Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GBC: Gradient-Based Connections for Optimizing Multi-Agent Systems\"", + "authors": "Xiaocheng Yang, Abdulrahman Alrabah, Dilek Hakkani-Tür, Gokhan Tur", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28187", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28270-agent-native-immune-system-architecture-taxonomy-and-engineering.md", + "title": "\"Agent-Native Immune System: Architecture, Taxonomy, and Engineering\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agent-Native Immune System: Architecture, Taxonomy, and Engineering\"", + "authors": "Bo Shen, Lifeng Chang, Tianyuan Wei, Yunpeng Li, Feng Shi, Yichen Han, Peijie Gao, Shiyi Kuang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28270", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28279-agentic-hardware-design-as-repository-level-code-evolution.md", + "title": "Agentic Hardware Design as Repository-Level Code Evolution", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agentic Hardware Design as Repository-Level Code Evolution", + "authors": "Cunxi Yu, Chenhui Deng, Nathaniel Pinckney, Brucek Khailany", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28279", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28349-hmars-a-hierarchical-multi-agent-memory-system-for-long-context-reasoning.md", + "title": "\"HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning\"", + "authors": "Zeju Li, Ziyang Zheng, Yizhou Zhou, Qiang Xu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28349", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-03", + "updated_at": "2026-06-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28360-carolina-guide-a-multi-agent-rag-system-with-institutional-guardrails-for-academ.md", + "title": "\"Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance\"", + "authors": "Ben Torsion, Jun Zhou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28360", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-11", + "updated_at": "2026-06-11", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28374-recursive-self-evolving-agents-via-held-out-selection.md", + "title": "Recursive Self-Evolving Agents via Held-Out Selection", + "type": "paper", + "meta": { + "type": "paper", + "title": "Recursive Self-Evolving Agents via Held-Out Selection", + "authors": "Michael Nguyen, Quoc Nguyen, Paul Vuong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28374", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-17", + "updated_at": "2026-06-17", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28409-evidence-driven-llm-agent-for-c-to-synthesizable-c-conversion-and-verification.md", + "title": "Evidence-Driven LLM Agent for C-to-Synthesizable-C Conversion and Verification", + "type": "paper", + "meta": { + "type": "paper", + "title": "Evidence-Driven LLM Agent for C-to-Synthesizable-C Conversion and Verification", + "authors": "Zhe Zhao, Hongbing Lang, Zhihan Xiao, Luke Ztz Hu, John Imoleayo Adebisi, Songping Mai", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28409", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "rag", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28425-tool-use-enables-undetectable-steganography-in-multi-agent-llm-systems.md", + "title": "Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems", + "authors": "Jimmy Laurence Rippin, Simon C. Marshall, David Demitri Africa, Christian Schroeder de Witt", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28425", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-25", + "updated_at": "2026-06-25", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "25", + "collection_queries": "agentic-ai, ai-agent, autonomous-agent-llm, multi-agent-llm, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28430-building-to-the-test-coding-agents-deliver-what-you-check-not-what-you-requested.md", + "title": "\"Building to the Test: Coding Agents Deliver What You Check, Not What You Requested\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Building to the Test: Coding Agents Deliver What You Check, Not What You Requested\"", + "authors": "Yanuo Ma, Ben Kereopa-Yorke, Ben Schultz", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28430", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28434-swe-mem-learning-adaptive-memory-management-for-long-horizon-coding-agents.md", + "title": "\"SWE-MeM: Learning Adaptive Memory Management for Long-Horizon Coding Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SWE-MeM: Learning Adaptive Memory Management for Long-Horizon Coding Agents\"", + "authors": "Shuzheng Gao, Wenhao Zeng, Zhaojian Yu, Jianqiao Wangni, Chaozheng Wang, Kai Cai, Shilin He, Michael R. Lyu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28434", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "memory", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28436-dockerless-environment-free-program-verifier-for-coding-agents.md", + "title": "\"Dockerless: Environment-Free Program Verifier for Coding Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Dockerless: Environment-Free Program Verifier for Coding Agents\"", + "authors": "Wenhao Zeng, Yuling Shi, Xiaodong Gu, Chao Hu, Chaofan Wang, Yuhao Cui, Hongting Zhou, Mengnan Qi, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28436", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28450-llm-agents-security-duality-a-comprehensive-survey-of-self-security-and-empowere.md", + "title": "\"LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity\"", + "authors": "Yiwei Xu, Yong Zhuang, Xuanming Liu, Tian Zhang, Bowen Xiao, Xiaoyang Xu, Delong Jiang, Juan Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28450", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-safety, llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28456-is-lying-an-emergent-behaviour-in-llms-evidence-from-gaslighting-ai-agents-in-a-.md", + "title": "Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game", + "type": "paper", + "meta": { + "type": "paper", + "title": "Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game", + "authors": "Subhendu Bhandary, Federico Carucci, Christos Charalambous, Francesca Dilisante, Ksenia Dvorkina, Anna Garbo, Jiaqi Liang, Riccardo Vasellini, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28456", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "memory", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "ai-agent, llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28467-an-agentic-ai-pipeline-for-appliance-level-energy-anomaly-detection-and-llm-driv.md", + "title": "An Agentic AI Pipeline for Appliance-Level Energy Anomaly Detection and LLM-Driven Recommendations", + "type": "paper", + "meta": { + "type": "paper", + "title": "An Agentic AI Pipeline for Appliance-Level Energy Anomaly Detection and LLM-Driven Recommendations", + "authors": "Dihia Falouz, Aida Douaibia, Amine Bechar, Youssef Elmir, Abbes Amira, Adel Oulefki", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28467", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agentic-ai, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28480-tua-bench-a-benchmark-for-general-purpose-terminal-use-agents.md", + "title": "\"TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents\"", + "authors": "Shoufa Chen, Luyuan Wang, Xuan Yang, Zhiheng Liu, Yuren Cong, Yuanfeng Ji, Feiyan Zhou, Xiaohui Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28480", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28570-digitizing-coaching-intelligence-an-agentic-framework-for-holistic-athlete-profi.md", + "title": "\"Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG\"", + "authors": "Deep Ghosal, Ishani Sen, Wazib Ansar, Amlan Chakrabarti", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28570", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-26", + "updated_at": "2026-06-26", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28666-why-trust-your-agent-empirical-security-gains-from-trism-guided-agentic-workflow.md", + "title": "Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare", + "type": "paper", + "meta": { + "type": "paper", + "title": "Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare", + "authors": "Liam Kearns", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28666", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agentic-ai, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28679-capability-gates-are-not-authorization-confused-deputy-failures-in-llm-agent-fra.md", + "title": "\"Capability Gates Are Not Authorization: Confused-Deputy Failures in LLM Agent Frameworks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Capability Gates Are Not Authorization: Confused-Deputy Failures in LLM Agent Frameworks\"", + "authors": "David Mellafe Zuvic", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28679", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28692-an-ai-agent-for-treatment-reasoning-over-a-biomedical-tool-universe.md", + "title": "An AI agent for treatment reasoning over a biomedical tool universe", + "type": "paper", + "meta": { + "type": "paper", + "title": "An AI agent for treatment reasoning over a biomedical tool universe", + "authors": "Shanghua Gao, Ayush Noori, Richard Zhu, Curtis Ginder, Zhenglun Kong, Xiaorui Su, Justin Kauffman, Benjamin S. Glicksberg, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28692", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "ai-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28733-agentic-abstention-do-agents-know-when-to-stop-instead-of-act.md", + "title": "\"Agentic Abstention: Do Agents Know When to Stop Instead of Act?\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic Abstention: Do Agents Know When to Stop Instead of Act?\"", + "authors": "Han Luo, Bingbing Wen, Lucy Lu Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28733", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28739-agent-safety-is-action-alignment.md", + "title": "Agent Safety Is Action Alignment", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agent Safety Is Action Alignment", + "authors": "Shawn Li, Yue Zhao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28739", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28781-hyphaedb-a-living-knowledge-topology-for-agent-first-memory.md", + "title": "\"HyphaeDB: A Living Knowledge Topology for Agent-First Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HyphaeDB: A Living Knowledge Topology for Agent-First Memory\"", + "authors": "Krishna Halaharvi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28781", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "memory", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory, agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28791-from-determinism-to-delegation-ai-native-software-engineering-and-the-evolution-.md", + "title": "\"From Determinism to Delegation: AI-Native Software Engineering and the Evolution of the Agentic Engineer\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Determinism to Delegation: AI-Native Software Engineering and the Evolution of the Agentic Engineer\"", + "authors": "Mamdouh Alenezi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28791", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "23", + "collection_queries": "agentic-ai, autonomous-agent-llm, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28839-the-contagion-tensor-a-framework-for-measuring-output-distribution-coupling-in-m.md", + "title": "\"The Contagion Tensor: A Framework for Measuring Output-Distribution Coupling in Multi-Agent LLM Systems -- and Auditing the Claims It Enables\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Contagion Tensor: A Framework for Measuring Output-Distribution Coupling in Multi-Agent LLM Systems -- and Auditing the Claims It Enables\"", + "authors": "Zewen Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28839", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28841-lamp-lean-based-agentic-framework-with-mcp-and-proof-repair.md", + "title": "\"LAMP: Lean-based Agentic framework with MCP and Proof Repair\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LAMP: Lean-based Agentic framework with MCP and Proof Repair\"", + "authors": "Santhana Srinivasan R, Maithilee Patawar", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28841", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LO", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28896-a-task-driven-and-quality-assured-agent-framework-for-sar-data-generation.md", + "title": "A Task-Driven and Quality-Assured Agent Framework for SAR Data Generation", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Task-Driven and Quality-Assured Agent Framework for SAR Data Generation", + "authors": "Xuanting Wu, Fan Zhanga, Fei Ma, Ling Guan, Guochun Ma, Yongsheng Zhou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28896", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.IV", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28925-multi-agent-routing-as-set-valued-prediction-a-wildchat-benchmark-and-cost-aware.md", + "title": "\"Multi-Agent Routing as Set-Valued Prediction: A WildChat Benchmark and Cost-Aware Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Multi-Agent Routing as Set-Valued Prediction: A WildChat Benchmark and Cost-Aware Evaluation\"", + "authors": "Ananto Nayan Bala, Faisal Muhammad Shah", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28925", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.IR", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-28958-when-latent-agents-lie-kv-cache-integrity-in-multi-agent-llm-collaboration.md", + "title": "\"When Latent Agents Lie: KV-Cache Integrity in Multi-Agent LLM Collaboration\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When Latent Agents Lie: KV-Cache Integrity in Multi-Agent LLM Collaboration\"", + "authors": "Luís Brito, Carlos Baquero", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.28958", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "memory", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29014-customized-generative-ai-agent-for-transportation-engineering-practice-a-develop.md", + "title": "\"Customized Generative AI Agent for Transportation Engineering Practice: A Development and Continued Pre-training Guideline\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Customized Generative AI Agent for Transportation Engineering Practice: A Development and Continued Pre-training Guideline\"", + "authors": "Dianwei Chen, Yuan-Zheng Lei, Zifan Zhang, Yuchen Liu, Xianfeng Yang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29014", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "planning", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.DL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29026-preventing-error-propagation-in-multi-agent-ai-through-runtime-monitoring.md", + "title": "Preventing Error Propagation in Multi-Agent AI through Runtime Monitoring", + "type": "paper", + "meta": { + "type": "paper", + "title": "Preventing Error Propagation in Multi-Agent AI through Runtime Monitoring", + "authors": "Shahnewaz Karim Sakib, Anindya Bijoy Das", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29026", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.ET" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29030-memory-as-an-attack-surface-in-llm-agents-a-study-on-multiple-choice-question-an.md", + "title": "\"Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering\"", + "authors": "Shahnewaz Karim Sakib, Anindya Bijoy Das", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29030", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.ET" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "ai-agent, llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29116-characterizing-large-language-model-agentic-workflows-a-study-on-n8n-ecosystem.md", + "title": "\"Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem\"", + "authors": "Yutian Tang, Yuming Zhou, Huaming Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29116", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-27", + "updated_at": "2026-06-27", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agentic-ai, llm-agent, planning-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29142-agent-security-meets-regulatory-reality-a-practitioner-systematization-of-autono.md", + "title": "Agent Security Meets Regulatory Reality -- A Practitioner Systematization of Autonomous-Agent Threats and Controls in Regulated Financial Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agent Security Meets Regulatory Reality -- A Practitioner Systematization of Autonomous-Agent Threats and Controls in Regulated Financial Systems", + "authors": "Krishna Mohan, Guda Nagavenkata Srinivasa", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29142", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety, autonomous-agent-llm, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29178-selective-memory-retention-for-long-horizon-llm-agents.md", + "title": "Selective Memory Retention for Long-Horizon LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Selective Memory Retention for Long-Horizon LLM Agents", + "authors": "Pranath Reddy", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29178", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29193-a-multi-dataset-benchmark-for-evaluating-llm-agents-in-microservice-failure-diag.md", + "title": "A Multi-Dataset Benchmark for Evaluating LLM Agents in Microservice Failure Diagnosis", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Multi-Dataset Benchmark for Evaluating LLM Agents in Microservice Failure Diagnosis", + "authors": "Yuanhong Cai, Xiaohui Nie, Kanglin Yin, Changhua Pei, Yongqian Sun, Shenglin Zhang, Haibin Liu, Guiyang Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29193", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29225-policyguard-a-dialogue-grounded-sub-agent-verifier-for-policy-adherence-in-llm-a.md", + "title": "\"PolicyGuard: A Dialogue-Grounded Sub-Agent Verifier for Policy Adherence in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PolicyGuard: A Dialogue-Grounded Sub-Agent Verifier for Policy Adherence in LLM Agents\"", + "authors": "Seongjae Kang, Taehyung Yu, Sung Ju Hwang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29225", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29270-minority-sentinel-when-to-overturn-majority-voting-in-multi-agent-llm-debates.md", + "title": "\"Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates\"", + "authors": "Chuan He, Zebin Chen, Zhengyi Yang, Shaobo Qiao, Mingchen Ju, Jiate Liu, Dong Wen, Guanfeng Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29270", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29315-hierarchical-experimentalist-agents.md", + "title": "Hierarchical Experimentalist Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Hierarchical Experimentalist Agents", + "authors": "Abhranil Chandra, Sankaran Vaidyanathan, Utsav Dhanuka, Varun Gandhi, Scott Niekum", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29315", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29354-when-llms-develop-languages-symbolic-communication-for-efficient-multi-agent-rea.md", + "title": "\"When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning\"", + "authors": "Zhengqi Pei, Qingming Huang, Shuhui Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29354", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.NE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29445-bridging-videoqa-and-video-guided-agentic-tasks-via-generalized-keyframe-extract.md", + "title": "Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction", + "type": "paper", + "meta": { + "type": "paper", + "title": "Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction", + "authors": "Sunqi Fan, Qingle Liu, Runqi Yin, Meng-Hao Guo, Shuojin Yang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29445", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29495-cognitive-world-models-for-process-level-social-influence-evaluation.md", + "title": "Cognitive World Models for Process-Level Social Influence Evaluation", + "type": "paper", + "meta": { + "type": "paper", + "title": "Cognitive World Models for Process-Level Social Influence Evaluation", + "authors": "Minghui Ma, Bin Guo, Han Wang, Mengqi Chen, Jingqi Liu, Yan Liu, Zhiwen Yu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29495", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29537-osworld2-0-benchmarking-computer-use-agents-on-long-horizon-real-world-tasks.md", + "title": "\"OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks\"", + "authors": "Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29537", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29654-budgeted-act-or-defer-multi-agent-llm-deliberation-with-local-reliability-bounds.md", + "title": "Budgeted Act-or-Defer Multi-Agent LLM Deliberation with Local Reliability Bounds", + "type": "paper", + "meta": { + "type": "paper", + "title": "Budgeted Act-or-Defer Multi-Agent LLM Deliberation with Local Reliability Bounds", + "authors": "Mengdie Flora Wang, Haochen Xie, Guanghui Wang, Devin Zhang, Jae Oh Woo", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29654", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29719-a-diagnostic-framework-and-multi-evaluator-audit-of-evaluator-driven-preference-.md", + "title": "A Diagnostic Framework and Multi-Evaluator Audit of Evaluator-Driven Preference Dynamics in Self-Adapting LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Diagnostic Framework and Multi-Evaluator Audit of Evaluator-Driven Preference Dynamics in Self-Adapting LLM Agents", + "authors": "Liu Zewen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29719", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29742-microagent-context-augmented-multi-agent-framework-for-automatic-microservice-de.md", + "title": "\"MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition\"", + "authors": "Zishan Su, Junjie Huang, Shiwen Shan, Xingyan Chen, Hui Zeng, Yuxin Su, Yanlin Wang, Michael R. Lyu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29742", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29745-echo-learning-epistemically-adaptive-language-agents-with-turn-level-credit.md", + "title": "\"ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit\"", + "authors": "Abhijnan Nath, Nikhil Krishnaswamy", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29745", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29746-deepmed-search-an-open-source-agentic-platform-for-medical-deep-research-with-in.md", + "title": "\"DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification\"", + "authors": "Maolin Liu, Fanyu Xu, Ruoqing Xu, Jiahang Zhang, Hao Wang, Rui Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29746", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29762-do-recommendation-algorithms-work-when-users-are-llm-agents-a-case-study-on-molt.md", + "title": "Do Recommendation Algorithms Work When Users Are LLM Agents? A Case Study on Moltbook", + "type": "paper", + "meta": { + "type": "paper", + "title": "Do Recommendation Algorithms Work When Users Are LLM Agents? A Case Study on Moltbook", + "authors": "Daming Li, Simeng Han, Jialu Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29762", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "ai-agent, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29771-clqt-a-closed-loop-cost-aware-strategy-consistent-benchmark-for-diagnostic-evalu.md", + "title": "\"CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents\"", + "authors": "Bo Qu, Mingguang Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29771", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG", + "q-fin.CP", + "q-fin.PM" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29774-analytic-concept-centric-memory-for-agentic-embodied-manipulation.md", + "title": "Analytic Concept-Centric Memory for Agentic Embodied Manipulation", + "type": "paper", + "meta": { + "type": "paper", + "title": "Analytic Concept-Centric Memory for Agentic Embodied Manipulation", + "authors": "Mingyang Sun, Xiujian Liang, Jiude Wei, Qichen He, Donglin Wang, Cewu Lu, Jianhua Sun", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29774", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29778-mandol-an-agglomerative-agent-memory-system-for-long-term-conversations.md", + "title": "\"Mandol: An Agglomerative Agent Memory System for Long-Term Conversations\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Mandol: An Agglomerative Agent Memory System for Long-Term Conversations\"", + "authors": "Yuhan Zhang, Zhiyuan Guo, Ziheng Zeng, Wei Wang, Wentao Wu, Lijie Xu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29778", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DB", + "cs.AI", + "cs.CL", + "cs.IR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29788-memleak-diagnosing-information-leaks-in-multimodal-agent-memory.md", + "title": "\"MemLeak: Diagnosing Information Leaks in Multimodal Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemLeak: Diagnosing Information Leaks in Multimodal Agent Memory\"", + "authors": "Kuan Wang, Chao Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29788", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-memory, ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29824-neural-procedural-memory-empowering-llm-agents-with-implicit-activation-steering.md", + "title": "\"Neural Procedural Memory: Empowering LLM Agents with Implicit Activation Steering\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Neural Procedural Memory: Empowering LLM Agents with Implicit Activation Steering\"", + "authors": "Chengfeng Zhao, Yuqiao Tan, Shizhu He, Yequan Wang, Jun Zhao, Kang Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29824", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "24", + "collection_queries": "agent-evaluation, agent-memory, autonomous-agent-llm, llm-agent, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29894-saber-math-automated-benchmark-for-information-retrieval-evaluation-in-mathemati.md", + "title": "\"SABER-Math: Automated Benchmark for Information Retrieval Evaluation in Mathematics\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SABER-Math: Automated Benchmark for Information Retrieval Evaluation in Mathematics\"", + "authors": "Nikolay Georgiev, Maria Drencheva, Kseniia Ibragimova, Ivo Petrov, Dimitar I. Dimitrov, Martin Vechev", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29894", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI", + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29914-memdelta-controlled-baselines-and-hidden-confounds-in-agent-memory-evaluation.md", + "title": "\"MemDelta: Controlled Baselines and Hidden Confounds in Agent Memory Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemDelta: Controlled Baselines and Hidden Confounds in Agent Memory Evaluation\"", + "authors": "Kuan Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29914", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29932-saga-scene-aware-goal-evolving-agents-for-long-horizon-civrealm-strategy-plannin.md", + "title": "\"SAGA: Scene-Aware, Goal-Evolving Agents for Long-Horizon CivRealm Strategy Planning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SAGA: Scene-Aware, Goal-Evolving Agents for Long-Horizon CivRealm Strategy Planning\"", + "authors": "Tianyu Jin, Shuo Chen, Yida Wang, Liuyu Xiang, Yingzhuo Liu, Zhiyao Jiang, Yexin Li, Zhaofeng He", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29932", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29957-swe-together-evaluating-coding-agents-in-interactive-user-sessions.md", + "title": "\"SWE-Together: Evaluating Coding Agents in Interactive User Sessions\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SWE-Together: Evaluating Coding Agents in Interactive User Sessions\"", + "authors": "Yifan Wu, Zhuokai Zhao, Songlin Li, Ho Hin Lee, Jiacheng Zhu, Shirley Wu, Tianhe Yu, Serena Li, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29957", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-evaluation, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-29961-duomem-towards-capable-on-device-memory-agents-via-dual-space-distillation.md", + "title": "\"DuoMem: Towards Capable On-Device Memory Agents via Dual-Space Distillation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DuoMem: Towards Capable On-Device Memory Agents via Dual-Space Distillation\"", + "authors": "Peyman Hosseini, Ondrej Bohdal, Ahmed Alajrami, Andrea Maracani, Ignacio Castro, Matthew Purver, Mete Ozay, Savas Ozkan, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.29961", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30005-llm-agents-are-latent-context-managers-eliciting-self-managed-context-via-a-prop.md", + "title": "\"LLM Agents Are Latent Context Managers: Eliciting Self-Managed Context via a Proprioceptive Dashboard\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LLM Agents Are Latent Context Managers: Eliciting Self-Managed Context via a Proprioceptive Dashboard\"", + "authors": "Binyan Xu, Haitao Li, Kehuan Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30005", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30111-automating-the-design-of-embodied-agent-architectures.md", + "title": "Automating the Design of Embodied Agent Architectures", + "type": "paper", + "meta": { + "type": "paper", + "title": "Automating the Design of Embodied Agent Architectures", + "authors": "Jian Zhou, Sihao Lin, Jin Li, Shuai Fu, Gengze Zhou, Qi Wu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30111", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30119-on-the-internet-nobody-knows-you-re-an-llm-bot-unmasking-web-agents-with-multi-l.md", + "title": "\"On the Internet, Nobody Knows You're an LLM Bot: Unmasking Web Agents with Multi-Layer Fingerprinting\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"On the Internet, Nobody Knows You're an LLM Bot: Unmasking Web Agents with Multi-Layer Fingerprinting\"", + "authors": "Iliana Fayolle, Sihem Bouhenniche, Samuel Pélissier, Pierre Laperdrix, Clémentine Maurice, Walter Rudametkin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30119", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "embodied-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30185-dynamo-dynamic-skill-tool-evolution-for-vision-language-agents.md", + "title": "\"Dynamo: Dynamic Skill-Tool Evolution for Vision-Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Dynamo: Dynamic Skill-Tool Evolution for Vision-Language Agents\"", + "authors": "Yutao Sun, Yanting Miao, Hao-Xuan Ma, Mengyu Zhou, Mingshuai Chen, Tiancheng Zhao, Dexin Wang, Lei Lv, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30185", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30251-taco-tool-augmented-credit-optimization-for-agentic-tool-use.md", + "title": "\"TACO: Tool-Augmented Credit Optimization for Agentic Tool Use\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TACO: Tool-Augmented Credit Optimization for Agentic Tool Use\"", + "authors": "Mingkuan Feng, Jinyang Wu, Hao Gu, Fangrui Lv, Ruihan Jin, Chuyuan Zhang, Zhengqi Wen, Jianhua Tao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30251", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30259-multi-agentic-system-leveraging-open-source-llms-to-mitigate-disinformation-thre.md", + "title": "Multi-Agentic System Leveraging Open-Source LLMs to Mitigate Disinformation Threats", + "type": "paper", + "meta": { + "type": "paper", + "title": "Multi-Agentic System Leveraging Open-Source LLMs to Mitigate Disinformation Threats", + "authors": "Sebastian Kula, Martin Tamajka", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30259", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30266-towards-continual-motion-language-agents-lora-variants-for-incremental-motion-un.md", + "title": "\"Towards Continual Motion-Language Agents: LoRA Variants for Incremental Motion Understanding and Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Towards Continual Motion-Language Agents: LoRA Variants for Incremental Motion Understanding and Generation\"", + "authors": "Bertram Taetz, Hugo Albuquerque Cosme da Silva, Gabriele Bleser-Taetz", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30266", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm, language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30294-rehearsed-multi-agent-live-product-demonstrations-with-real-time-voice-question-.md", + "title": "Rehearsed Multi-Agent Live Product Demonstrations with Real-Time Voice Question Answering", + "type": "paper", + "meta": { + "type": "paper", + "title": "Rehearsed Multi-Agent Live Product Demonstrations with Real-Time Voice Question Answering", + "authors": "Rahul Khedar, Mayank Malhotra, Avinash Karn, Mouli V, Prakhar Mehrotra", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30294", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.HC", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30383-whose-side-is-your-agent-on-multi-party-principal-loyalty-in-llm-agents.md", + "title": "Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents", + "authors": "Bojie Li, Noah Shi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30383", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30454-collective-cooperation-without-individual-fidelity-in-llm-agents.md", + "title": "Collective cooperation without individual fidelity in LLM agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Collective cooperation without individual fidelity in LLM agents", + "authors": "Henrique Ferraz de Arruda, Carlos Gracia Lázaro, Alberto Aleta, Yamir Moreno", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30454", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "physics.soc-ph", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30524-the-illusion-of-agentic-complexity-in-readme-md-generation-evaluating-single-age.md", + "title": "\"The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems\"", + "authors": "Abu Saleh, Tesfay Welegebreal Tesfay, Phuong T. Nguyen, Juri Di Rocco, Muhammad Umar Zeshan, Davide Di Ruscio", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30524", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "multi-agent-llm, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30546-mas-lab-a-specification-driven-validation-framework-for-reliable-multi-agent-sys.md", + "title": "\"MAS-Lab: A Specification-Driven Validation Framework for Reliable Multi-Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MAS-Lab: A Specification-Driven Validation Framework for Reliable Multi-Agent Systems\"", + "authors": "Jordan Augé, Giovanna Carofiglio, Giulio Grassi, Jacques Samain", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30546", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30555-linguistic-firewall-geometry-as-defense-in-multi-agent-systems-routing.md", + "title": "\"Linguistic Firewall: Geometry as Defense in Multi-Agent Systems Routing\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Linguistic Firewall: Geometry as Defense in Multi-Agent Systems Routing\"", + "authors": "Dvir Alsheich, Adar Peleg, Ben Hagag, Rom Himelstein, Amit Levi, Avi Mendelson", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30555", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30560-tracelab-characterizing-coding-agent-workloads-for-llm-serving.md", + "title": "\"TraceLab: Characterizing Coding Agent Workloads for LLM Serving\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TraceLab: Characterizing Coding Agent Workloads for LLM Serving\"", + "authors": "Kan Zhu, Mathew Jacob, Chenxi Ma, Yi Pan, Stephanie Wang, Arvind Krishnamurthy, Baris Kasikci", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30560", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.PF" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30566-forensic-trajectory-signatures-for-agent-memory-poisoning-detection.md", + "title": "Forensic Trajectory Signatures for Agent Memory Poisoning Detection", + "type": "paper", + "meta": { + "type": "paper", + "title": "Forensic Trajectory Signatures for Agent Memory Poisoning Detection", + "authors": "Jun Wen Leong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30566", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30573-swe-interact-reimagining-swe-benchmarks-as-user-driven-long-horizon-coding-sessi.md", + "title": "\"SWE-INTERACT: Reimagining SWE Benchmarks as User-Driven Long-Horizon Coding Sessions\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SWE-INTERACT: Reimagining SWE Benchmarks as User-Driven Long-Horizon Coding Sessions\"", + "authors": "Mohit Raghavendra, Anisha Gunjal, Aakash Sabharwal, Yunzhong He", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30573", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30602-mesa-prioritizing-vulnerable-communication-channels-for-securing-multi-agent-sys.md", + "title": "\"MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems\"", + "authors": "Kunyang Li, Kyle Domico, Jonathan Gregory, Patrick McDaniel", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30602", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30616-scaling-the-horizon-not-the-parameters-reaching-trillion-parameter-performance-w.md", + "title": "\"Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent\"", + "authors": "Lei Bai, Zongsheng Cao, Yang Chen, Zhiyao Cui, Shangheng Du, Yue Fan, Shiyang Feng, Zijie Guo, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30616", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30639-self-evolving-world-models-for-llm-agent-planning.md", + "title": "Self-Evolving World Models for LLM Agent Planning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Self-Evolving World Models for LLM Agent Planning", + "authors": "Xuan Zhang, Wenxuan Zhang, See-Kiong Ng, Yang Deng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30639", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "llm-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30697-lumos-a-semantic-operating-system-layer-for-accessibility-grounded-ai-agents.md", + "title": "\"LUMOS: A Semantic Operating-System Layer for Accessibility-Grounded AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LUMOS: A Semantic Operating-System Layer for Accessibility-Grounded AI Agents\"", + "authors": "Yogeswar Reddy Thota", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30697", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.OS", + "cs.AI", + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "ai-agent, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30755-understanding-and-evaluating-claw-like-agent-security-through-a-computer-systems.md", + "title": "Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens", + "type": "paper", + "meta": { + "type": "paper", + "title": "Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens", + "authors": "Peizhi Niu, Wenjie Qu, Shangding Gu, Tianneng Shi, Yuankai Li, Ahmad Tawaha, Hend Alzahrani, Vincent Siu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30755", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety, ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30840-contrastive-reflection-for-iterative-prompt-optimization.md", + "title": "Contrastive Reflection for Iterative Prompt Optimization", + "type": "paper", + "meta": { + "type": "paper", + "title": "Contrastive Reflection for Iterative Prompt Optimization", + "authors": "Derek Koh, Jinghui Mo, Benjamin H. Le, Jiening Zhan, Baofen Zheng, Kevin Bevis, Nathaniel C. Owen, Lauren Elizabeth Charney, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30840", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "ai-agent, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30877-a-systematic-approach-to-multi-agent-ai-from-advanced-regulatory-control-theory-.md", + "title": "\"A Systematic Approach to Multi-Agent AI from Advanced Regulatory Control Theory: Safe and Auditable LLM Operator Agents for Process Control\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"A Systematic Approach to Multi-Agent AI from Advanced Regulatory Control Theory: Safe and Auditable LLM Operator Agents for Process Control\"", + "authors": "Idelfonso B. R. Nogueira, Sigurd Skogestad", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30877", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30887-training-therapeutic-judges-and-multi-agent-systems-for-human-aligned-mental-hea.md", + "title": "Training Therapeutic Judges and Multi-Agent Systems for Human-Aligned Mental Health Support", + "type": "paper", + "meta": { + "type": "paper", + "title": "Training Therapeutic Judges and Multi-Agent Systems for Human-Aligned Mental Health Support", + "authors": "Mizanur Rahman, Abeer Badawi, Elahe Rahimi, Laleh Seyyed-Kalantari, Frank Rudzicz, Enamul Hoque, Elham Dolatabadi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30887", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30906-investigating-multi-agent-deliberation-in-law.md", + "title": "Investigating Multi-Agent Deliberation in Law", + "type": "paper", + "meta": { + "type": "paper", + "title": "Investigating Multi-Agent Deliberation in Law", + "authors": "Cor Steging, Ludi van Leeuwen, Tadeusz Zbiegień", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30906", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agentic-ai, ai-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30949-agrefactor-self-evolving-agentic-workflow-for-hls-compatibility-and-performance.md", + "title": "\"AgRefactor: Self-Evolving Agentic Workflow for HLS Compatibility and Performance\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgRefactor: Self-Evolving Agentic Workflow for HLS Compatibility and Performance\"", + "authors": "Yang Zou, Zijian Ding, Yizhou Sun, Jason Cong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30949", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.AR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agentic-ai, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30970-behavioral-governance-for-autonomous-ai-agents-the-agentbound-framework.md", + "title": "\"Behavioral Governance for Autonomous AI Agents: The AgentBound Framework\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Behavioral Governance for Autonomous AI Agents: The AgentBound Framework\"", + "authors": "Anuj Kaul, Qianlong Lan, Pranay Gupta", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30970", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-30986-the-organizational-behavior-of-agentic-ai-collective-intelligence-in-human-agent.md", + "title": "\"The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows\"", + "authors": "Canhui Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.30986", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "planning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY", + "cs.HC", + "cs.MA", + "econ.GN" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agentic-ai, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31046-openlife-toward-open-world-artificial-life-with-autonomous-llm-agents.md", + "title": "\"OpenLife: Toward Open-World Artificial Life with Autonomous LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OpenLife: Toward Open-World Artificial Life with Autonomous LLM Agents\"", + "authors": "Atsushi Masumori, Itsuki Doi, Norihiro Maruyama, Ryosuke Takata, Takashi Ikegami", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31046", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31073-multiuav-plat-an-llm-oriented-platform-benchmark-and-framework-for-multi-uav-col.md", + "title": "\"MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning\"", + "authors": "Sheng Zhang, Qinglin Li, Yuechao Zang, Xueqin Huang, Yijia Fu, Cheng Zhu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31073", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA", + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-evaluation, llm-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31085-ddiagents-mechanism-conditioned-context-flow-for-drug-drug-interaction-predictio.md", + "title": "\"DDIAgents: Mechanism-Conditioned Context Flow for Drug-Drug Interaction Prediction\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DDIAgents: Mechanism-Conditioned Context Flow for Drug-Drug Interaction Prediction\"", + "authors": "Zhenqian Shen, Yu Liu, Xiaoyi Fu, Quanming Yao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31085", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31134-beyond-the-library-an-agentic-framework-for-autoformalizing-research-mathematics.md", + "title": "\"Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics\"", + "authors": "Arshia Soltani Moakhar, Iman Gholami, Max Springer, Mahdi JafariRaviz, MohammadTaghi Hajiaghayi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31134", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31174-clawarena-team-benchmarking-subagent-orchestration-and-dynamic-workflows-in-lang.md", + "title": "\"ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents\"", + "authors": "Kaiwen Xiong, Haonian Ji, Shi Qiu, Zeyu Zheng, Cihang Xie, Xinyu Ye, Huaxiu Yao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31174", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31179-healthagentbench-a-unified-benchmark-suite-of-realistic-agentic-healthcare-envir.md", + "title": "\"HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents\"", + "authors": "Qianchu Liu, Sheng Zhang, Guanghui Qin, Jeya Maria Jose Valanarasu, Maximilian Rokuss, Mingyu Lu, Timothy Ossowski, Juan Manuel Zambrano Chaves, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31179", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-evaluation, ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31200-agentic-rag-vlm-affordance-aware-retrieval-augmented-generation-with-self-reflec.md", + "title": "\"Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping\"", + "authors": "Tao Chen, Lizheng Liu, Jiaxu Wang, Ziyue Jiang, Ruiqi Tian, JiGuang Huo, Zhongxue Gan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31200", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31209-long-term-traffic-simulation-via-structured-autoregressive-modeling.md", + "title": "Long-term Traffic Simulation via Structured Autoregressive Modeling", + "type": "paper", + "meta": { + "type": "paper", + "title": "Long-term Traffic Simulation via Structured Autoregressive Modeling", + "authors": "Lingyu Xiao, Zexin Feng, Xintao Yan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31209", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "planning", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31227-securing-the-ai-agent-a-unified-framework-for-multi-layer-agent-red-teaming.md", + "title": "\"Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming\"", + "authors": "Yong Yang, Xing Zheng, Huiyu Wu, Huangsheng Cheng, Xiaorong Shi, Jing Guo, Bo Yang, Yi Zhou, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31227", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-safety, ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31229-agentic-ideation-sample-efficient-agentic-trajectories-synthesis-for-scientific-.md", + "title": "\"Agentic-Ideation: Sample Efficient Agentic Trajectories Synthesis for Scientific Ideation Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic-Ideation: Sample Efficient Agentic Trajectories Synthesis for Scientific Ideation Agents\"", + "authors": "Keyu Zhao, Lingyan Kong, Fengli Xu, Yong Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31229", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agentic-ai, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31252-embodied-cad-solver-grounded-llm-agents-for-parametric-b-rep-assembly-modeling.md", + "title": "\"Embodied CAD: Solver-Grounded LLM Agents for Parametric B-Rep Assembly Modeling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Embodied CAD: Solver-Grounded LLM Agents for Parametric B-Rep Assembly Modeling\"", + "authors": "Fumin Liu, Haoyu Zhou, Fei Hao, Lin Yang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31252", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31314-a-novel-method-for-differential-algebraic-dynamic-model-discovery-in-power-syste.md", + "title": "\"A Novel Method for Differential-Algebraic Dynamic Model Discovery in Power Systems: An LLM-Based Multi-Agent Collaborative Framework\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"A Novel Method for Differential-Algebraic Dynamic Model Discovery in Power Systems: An LLM-Based Multi-Agent Collaborative Framework\"", + "authors": "Xinming Wang, Fan Tang, Yingli Wei, Yakun He, Zhe Liu, Ping Jiang, Haoyu Wu, Zihan Guo, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31314", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31339-verification-gated-agentic-mission-state-governance-for-intelligent-industrial-m.md", + "title": "Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems", + "type": "paper", + "meta": { + "type": "paper", + "title": "Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems", + "authors": "Guoqin Tang, Qingxuan Jia, Yichen Tan, Zeyuan Huang, Ning Ji, Gang Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31339", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31410-xiaomi-gui-0-technical-report.md", + "title": "Xiaomi-GUI-0 Technical Report", + "type": "paper", + "meta": { + "type": "paper", + "title": "Xiaomi-GUI-0 Technical Report", + "authors": "Wanxia Cao, Chengzhen Duan, Pei Fu, Pengzhi Gao, Niu Lian, Fazhan Liu, Hui Liu, Heng Qu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31410", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31471-think-while-you-map-asynchronous-vision-language-agents-for-incremental-3d-scene.md", + "title": "\"Think While You Map: Asynchronous Vision-Language Agents for Incremental 3D Scene Graphs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Think While You Map: Asynchronous Vision-Language Agents for Incremental 3D Scene Graphs\"", + "authors": "Deniz Bickici, Michael Pabst, Shohei Mori, Dieter Schmalstieg", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31471", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31612-what-memory-do-gui-agents-really-need-from-passive-records-to-active-task-drivin.md", + "title": "What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States", + "type": "paper", + "meta": { + "type": "paper", + "title": "What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States", + "authors": "Chen Liu, Ling Chen, Hanzhang Zhou, Xu Zhang, Quyu Kong, Panrong Tong, Wenhao Wang, Xin Yu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31612", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-memory, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31635-a-tutorial-on-autonomous-fault-tolerant-control-using-knowledge-grounded-llm-age.md", + "title": "A Tutorial on Autonomous Fault-Tolerant Control Using Knowledge-Grounded LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Tutorial on Autonomous Fault-Tolerant Control Using Knowledge-Grounded LLM Agents", + "authors": "Javal Vyas, Milapji Singh Gill, Artan Markaj, Felix Gehlhoff, Mehmet Mercangöz", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31635", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31639-a-lifecycle-and-application-stack-survey-of-large-language-model-vulnerabilities.md", + "title": "\"A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems\"", + "authors": "Seyed Bagher Hashemi Natanzi, Bo Tang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31639", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "embodied-agent", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.GT", + "cs.LO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation, autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31648-think-in-english-answer-in-korean-efficient-adaptation-of-multilingual-tool-usin.md", + "title": "\"Think in English, Answer in Korean: Efficient Adaptation of Multilingual Tool-Using Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Think in English, Answer in Korean: Efficient Adaptation of Multilingual Tool-Using Agents\"", + "authors": "Utsav Garg, Sungjin Hong, Jason Jung, Justin Lee, Shaan Desai, Joon Hee Kim, Anirudh Shrinivason, Edmond Wen, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31648", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "multi-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai, function-calling, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31650-echo-prune-to-act-trace-to-learn-with-selective-turn-memory-in-agentic-rl.md", + "title": "\"ECHO: Prune to act, trace to learn with selective turn memory in agentic RL\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ECHO: Prune to act, trace to learn with selective turn memory in agentic RL\"", + "authors": "Zijun Xie, Binbin Zheng, Enlei Gong, Jihua Liu, Yuyang You, Lingfeng Liu, Jiayao Tang, Guanqun Zhao, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31650", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31665-forecastagentsearch-towards-a-multi-expert-agent-search-system-for-geopolitical-.md", + "title": "\"ForecastAgentSearch: Towards a Multi-Expert Agent Search System for Geopolitical Event Forecasting\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ForecastAgentSearch: Towards a Multi-Expert Agent Search System for Geopolitical Event Forecasting\"", + "authors": "Miaomiao Cai, He Chang, Yunshan Ma, See-kiong Ng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31665", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31693-shopx-a-foundation-model-for-intent-to-item-fulfillment-in-agentic-shopping.md", + "title": "\"ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping\"", + "authors": "Jiacheng Chen, Tao Zhang, Manxi Lin, Dunxian Huang, Teng Shi, Honghao Fu, Mengyan Li, Xinming Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31693", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31744-a-conversational-agentic-interface-to-physics-based-household-digital-twins-for-.md", + "title": "A Conversational Agentic Interface to Physics-Based Household Digital Twins for Residential Energy Decision Support", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Conversational Agentic Interface to Physics-Based Household Digital Twins for Residential Energy Decision Support", + "authors": "Costas Mylonas, Titos Georgoulakis, Magda Foti", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31744", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31767-jeto-bench-a-reproducible-benchmark-for-execution-time-improvement-patches-in-ja.md", + "title": "\"JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java\"", + "authors": "Khashayar Etemadi, Zhendong Su", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31767", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31831-an-agentic-ai-framework-to-accelerate-scientific-discovery-in-plant-phenotyping.md", + "title": "An Agentic AI Framework to Accelerate Scientific Discovery in Plant Phenotyping", + "type": "paper", + "meta": { + "type": "paper", + "title": "An Agentic AI Framework to Accelerate Scientific Discovery in Plant Phenotyping", + "authors": "Renan Souza, Daniel Rosendo, Kelsey Carter, John Lagergren, Frédéric Suter, Shelaine L. Curd, Gerald A. Tuskan, Rafael Ferreira da Silva, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31831", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agentic-ai, ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31916-theory-of-mind-and-persuasion-beyond-conversation-assessing-the-capacity-of-llms.md", + "title": "\"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action\"", + "authors": "Ben Slater, Matteo G. Mecattaf, Lucy G. Cheke, John Burden, Winnie Street", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31916", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-31980-digitalcoach-communication-and-grounding-gaps-in-human-and-agentic-computer-use-.md", + "title": "\"DigitalCoach: Communication and Grounding Gaps in Human and Agentic Computer Use Coaching\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DigitalCoach: Communication and Grounding Gaps in Human and Agentic Computer Use Coaching\"", + "authors": "Meng Chen, Anya Ji, Tsung-Han Wu, Tobias Maringgele, David M. Chan, Alane Suhr, Amy Pavel", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.31980", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-32025-generative-skill-composition-for-llm-agents.md", + "title": "Generative Skill Composition for LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Generative Skill Composition for LLM Agents", + "authors": "Xinyu Zhao, Zhen Tan, Vaishnav Tadiparthi, Nakul Agarwal, Kwonjoon Lee, Ehsan Moradi Pari, Hossein Nourkhiz Mahjoub, Tianlong Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.32025", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "coding-agent, llm-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2606-32034-qval-cheaply-evaluating-dense-supervision-signals-for-long-horizon-llm-agents.md", + "title": "\"QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents\"", + "authors": "Sergio Hernández-Gutiérrez, Matteo Merler, Ilze Amanda Auzina, Joschka Strüber, Ameya Prabhu, Matthias Bethge", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2606.32034", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00038-stop-hand-holding-your-coding-agent-engineering-the-loops-that-replace-step-by-s.md", + "title": "\"Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting\"", + "authors": "Sandeco Macedo", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00038", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-28", + "updated_at": "2026-06-28", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00041-atm-cid-brokered-pre-write-admission-for-multi-agent-code-co-synthesis.md", + "title": "\"ATM: CID-Brokered Pre-Write Admission for Multi-Agent Code Co-Synthesis\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ATM: CID-Brokered Pre-Write Admission for Multi-Agent Code Co-Synthesis\"", + "authors": "Eagl Huang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00041", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-29", + "updated_at": "2026-06-29", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00233-from-signals-to-structure-how-memory-architecture-drives-language-emergence-in-l.md", + "title": "\"From Signals to Structure: How Memory Architecture Drives Language Emergence in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Signals to Structure: How Memory Architecture Drives Language Emergence in LLM Agents\"", + "authors": "Yashar Talebirad, Eden Redman, Ali Parsaee, Osmar R. Zaiane", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00233", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.IT", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00255-slm-llm-or-agentic-ai-toward-intelligent-uav-enabled-wpt-systems-in-low-altitude.md", + "title": "SLM, LLM or Agentic AI? Toward Intelligent UAV-Enabled WPT Systems in Low-Altitude Economy Networks", + "type": "paper", + "meta": { + "type": "paper", + "title": "SLM, LLM or Agentic AI? Toward Intelligent UAV-Enabled WPT Systems in Low-Altitude Economy Networks", + "authors": "Feibo Jiang, Li Dong, Lei Mao, Kezhi Wang, Xianbin Wang, Abbas Jamalipour", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00255", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "reasoning", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IT" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00297-epc-a-standardized-protocol-for-measuring-evaluator-preference-dynamics-in-llm-a.md", + "title": "\"EPC: A Standardized Protocol for Measuring Evaluator Preference Dynamics in LLM Agent Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EPC: A Standardized Protocol for Measuring Evaluator Preference Dynamics in LLM Agent Systems\"", + "authors": "Zewen Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00297", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00334-managed-autonomy-at-runtime-gear-based-safety-and-governance-for-single-and-mult.md", + "title": "\"Managed Autonomy at Runtime: Gear-Based Safety and Governance for Single- and Multi-Agent Cyber-Physical Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Managed Autonomy at Runtime: Gear-Based Safety and Governance for Single- and Multi-Agent Cyber-Physical Systems\"", + "authors": "Srini Ramaswamy, Wang Miaosheng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00334", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "embodied-agent", + "multi-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "autonomous-agent-llm, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00345-registry-governed-agent-lifecycle-completing-eddops-with-evaluation-drivenregist.md", + "title": "\"Registry-Governed Agent Lifecycle:Completing EDDOps with Evaluation-DrivenRegistration, Promotion, and Retirement on AWS AgentCore\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Registry-Governed Agent Lifecycle:Completing EDDOps with Evaluation-DrivenRegistration, Promotion, and Retirement on AWS AgentCore\"", + "authors": "Richard Kang, Vincent Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00345", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00407-personalization-as-inverse-planning-learning-latent-design-intents-for-agentic-s.md", + "title": "\"Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising\"", + "authors": "Tianci Liu, Zihan Dong, Linjun Zhang, Haoyu Wang, jing Gao, Emre Kiciman, Ranveer Chandra, Wei-Ting Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00407", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "multi-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00422-kidnaprag-a-black-box-attack-for-hijacking-reasoning-in-agentic-retrieval-augmen.md", + "title": "\"KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems\"", + "authors": "Chanwoo Choi, Euntae Kim, Kyuho Lee, Youngsam Chun, Jinhee Jeong, Eunmi Kim, Myunggyo Oh, Junseo Jang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00422", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00436-phreeqc-mcq-200-a-diagnostic-benchmark-for-tool-augmented-scientific-simulator-a.md", + "title": "\"PHREEQC-MCQ-200: A Diagnostic Benchmark for Tool-Augmented Scientific Simulator Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PHREEQC-MCQ-200: A Diagnostic Benchmark for Tool-Augmented Scientific Simulator Agents\"", + "authors": "Ke Zhang, Sahchit Chundur, Mohammad Javad Qomi, Maziar Raissi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00436", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00440-minos-a-multi-agent-collaborative-framework-for-provenance-based-backward-tracki.md", + "title": "\"Minos: A Multi-Agent Collaborative Framework for Provenance-Based Backward Tracking\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Minos: A Multi-Agent Collaborative Framework for Provenance-Based Backward Tracking\"", + "authors": "Jiahui Wang, Zhenyuan Li, Zhengkai Wang, Xiangmin Shen, Fan Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00440", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00454-agri-sage-simulation-grounded-multi-agent-llm-for-context-aware-agricultural-adv.md", + "title": "\"Agri-SAGE: Simulation-Grounded Multi-Agent LLM for Context-Aware Agricultural Advisory Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agri-SAGE: Simulation-Grounded Multi-Agent LLM for Context-Aware Agricultural Advisory Generation\"", + "authors": "Vedant Balasubramaniam, Geetha Charan, Manojkumar Patil, Rohit P Suresh, V Priyanka, Kodur Sai Vinay Sathvik, Y. Narahari", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00454", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00502-a-task-state-representation-for-long-horizon-mobile-gui-agents.md", + "title": "A Task-State Representation for Long-Horizon Mobile GUI Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Task-State Representation for Long-Horizon Mobile GUI Agents", + "authors": "Yujie Zheng, Zikang Liu, Xin Zhao, Ji-Rong Wen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00502", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00555-rise-from-the-ashes-llm-based-static-analysis-for-deep-learning-framework-bugs.md", + "title": "\"Rise From The Ashes: LLM-based Static Analysis for Deep Learning Framework Bugs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Rise From The Ashes: LLM-based Static Analysis for Deep Learning Framework Bugs\"", + "authors": "Shaoyu Yang, Haifeng Lin, Chunrong Fang, Xiang Chen, Wei Cheng, Jiawei Liu, Yiyu Zhang, Hongyu Liu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00555", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00604-vehicle-routing-problem-meets-large-language-models-an-overview-and-perspectives.md", + "title": "\"Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives\"", + "authors": "Xianchao Xiu, Chong Shen, Yanjiao Zhu, Wanquan Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00604", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "math.OC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00627-agi-maze-as-a-benchmark-framework-for-world-modeling-agents.md", + "title": "AGI Maze as a Benchmark Framework for World-Modeling Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "AGI Maze as a Benchmark Framework for World-Modeling Agents", + "authors": "Alexey Potapov", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00627", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00692-self-gc-self-governing-context-for-long-horizon-llm-agents.md", + "title": "\"Self-GC: Self-Governing Context for Long-Horizon LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Self-GC: Self-Governing Context for Long-Horizon LLM Agents\"", + "authors": "Xubin Hao, Hongjin Meng, Xin Yin, Jiawei Zhu, Chenpeng Cao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00692", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00911-from-registry-to-repository-how-ai-agent-skills-are-written-adapted-and-maintain.md", + "title": "\"From Registry to Repository: How AI Agent Skills Are Written, Adapted, and Maintained\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Registry to Repository: How AI Agent Skills Are Written, Adapted, and Maintained\"", + "authors": "Haoyu Gao, Jai Lal Lulla, Hong Yi Lin, Sebastian Baltes, Christoph Treude, Mansooreh Zahedi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00911", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "ai-agent, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00918-from-personas-to-plot-character-grounded-multi-agent-story-generation-for-long-f.md", + "title": "\"From Personas to Plot: Character-Grounded Multi-Agent Story Generation for Long-Form Narratives\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Personas to Plot: Character-Grounded Multi-Agent Story Generation for Long-Form Narratives\"", + "authors": "Aayush Aluru, Chloe Ho, Muhammad Hammouri, Kerry Luo, Myra Malik, Ryan Lagasse, Arjun Bahuguna, Vasu Sharma", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00918", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00939-leveraging-llm-based-agentic-systems-to-generate-quantum-applications-for-test-o.md", + "title": "Leveraging LLM-Based Agentic Systems to Generate Quantum Applications for Test Optimization", + "type": "paper", + "meta": { + "type": "paper", + "title": "Leveraging LLM-Based Agentic Systems to Generate Quantum Applications for Test Optimization", + "authors": "Ming Tao, Yuechen Li, Tao Yue, Man Zhang, Aitor Arrieta Marcos", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00939", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "quant-ph" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00972-bayesian-uncertainty-propagation-for-agentic-rag-pipelines-a-proof-of-concept-st.md", + "title": "\"Bayesian Uncertainty Propagation for Agentic RAG Pipelines: A Proof-of-Concept Study on Multi-Hop Question Answering\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Bayesian Uncertainty Propagation for Agentic RAG Pipelines: A Proof-of-Concept Study on Multi-Hop Question Answering\"", + "authors": "Louis Donaldson, Connor Walker, Koorosh Aslansefat, Yiannis Papadopoulos", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00972", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-00990-swe-doctor-guiding-software-engineering-agents-with-runtime-diagnosis-from-multi.md", + "title": "\"SWE-Doctor: Guiding Software Engineering Agents with Runtime Diagnosis from Multi-Faceted Bug Reproduction Tests\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SWE-Doctor: Guiding Software Engineering Agents with Runtime Diagnosis from Multi-Faceted Bug Reproduction Tests\"", + "authors": "Yaoqi Guo, Yang Liu, Jie M. Zhang, Yun Ma, Yiling Lou, Zhenpeng Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.00990", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01047-conversable-complexity-agentic-llm-collectives-as-interpretable-substrates.md", + "title": "\"Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates\"", + "authors": "Elias Najarro, Ane Espeseth, Eleni Nisioti, Sebastian Risi, Stefano Nichele", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01047", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01061-agentic-generation-of-verifiable-rules-for-deterministic-self-expanding-reaction.md", + "title": "Agentic generation of verifiable rules for deterministic, self-expanding reaction classification", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agentic generation of verifiable rules for deterministic, self-expanding reaction classification", + "authors": "Daniel Armstrong, Maarten Dobbelaere, Valentas Olikauskas, Helena Avila, Octavian Susanu, Jérôme Waser, Philippe Schwaller", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01061", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "multi-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01071-memsyco-bench-benchmarking-sycophancy-in-agent-memory.md", + "title": "\"MemSyco-Bench: Benchmarking Sycophancy in Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MemSyco-Bench: Benchmarking Sycophancy in Agent Memory\"", + "authors": "Zhishang Xiang, Zerui Chen, Yunbo Tang, Zhimin Wei, Ruqin Ning, Yujie Lin, Qinggang Zhang, Jinsong Su", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01071", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01084-can-agents-generalize-to-the-open-world-unveiling-the-fragility-of-static-traini.md", + "title": "Can Agents Generalize to the Open World? Unveiling the Fragility of Static Training in Tool Use", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can Agents Generalize to the Open World? Unveiling the Fragility of Static Training in Tool Use", + "authors": "Song-Lin Lv, Weiming Wu, Rui Zhu, Zi-Jian Cheng, Lan-Zhe Guo", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01084", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01120-next-generation-agentic-reinforcement-learning-systems-enable-self-evolving-agen.md", + "title": "Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents", + "authors": "Ran Yan, Wei Fu, Jiale Li, Shusheng Xu, Zhiyu Mei, Jiaxuan Gao, Jiarui Zhang, Wentai Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01120", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01211-are-performance-optimization-benchmarks-reliably-measuring-coding-agents.md", + "title": "Are Performance-Optimization Benchmarks Reliably Measuring Coding Agents?", + "type": "paper", + "meta": { + "type": "paper", + "title": "Are Performance-Optimization Benchmarks Reliably Measuring Coding Agents?", + "authors": "Zhi Chen, Zhensu Sun, Yuling Shi, David Lo, Lingxiao Jiang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01211", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01213-reporescue-an-empirical-study-of-llm-agents-on-whole-repository-compatibility-re.md", + "title": "\"RepoRescue: An Empirical Study of LLM Agents on Whole-Repository Compatibility Rescue\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RepoRescue: An Empirical Study of LLM Agents on Whole-Repository Compatibility Rescue\"", + "authors": "Zhihao Lin, Mingyi Zhou, Zhensu Sun, Yizhuo Yang, Renyu Yang, David Lo, Li Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01213", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01366-auto-fl-research-agentic-search-for-federated-learning-algorithms.md", + "title": "\"Auto-FL-Research: Agentic Search for Federated Learning Algorithms\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Auto-FL-Research: Agentic Search for Federated Learning Algorithms\"", + "authors": "Holger R. Roth, Ziyue Xu, Chester Chen, Daguang Xu, Peter Cnudde, Andrew Feng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01366", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agentic-ai, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01421-risk-architecture-for-ai-native-engineering-teams-an-organizational-framework-fo.md", + "title": "\"Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance\"", + "authors": "Laxmipriya Ganesh Iyer", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01421", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01510-janus-a-playground-for-user-involved-agentic-permission-management.md", + "title": "\"Janus: a Playground for User-Involved Agentic Permission Management\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Janus: a Playground for User-Involved Agentic Permission Management\"", + "authors": "Natalie Grace Brigham, Eugene Bagdasarian, Tadayoshi Kohno, Franziska Roesner", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01510", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01523-multi-head-recurrent-memory-agents.md", + "title": "Multi-Head Recurrent Memory Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Multi-Head Recurrent Memory Agents", + "authors": "Jiatong Li, Samuel Yeh, Sharon Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01523", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01531-opine-world-programmatic-world-modeling-with-ontology-error-prioritized-interact.md", + "title": "\"OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration\"", + "authors": "David Courtis, Wenhao Li, Scott Sanner", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01531", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "llm-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01600-boundary-sync-measuring-communication-induced-representational-coupling-in-multi.md", + "title": "\"BOUNDARY_SYNC: Measuring Communication-Induced Representational Coupling in Multi-Agent LLM Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"BOUNDARY_SYNC: Measuring Communication-Induced Representational Coupling in Multi-Agent LLM Systems\"", + "authors": "Zewen Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01600", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01640-agentflow-building-agent-dependency-graphs-for-static-analysis-of-agent-programs.md", + "title": "\"AgentFlow: Building Agent Dependency Graphs for Static Analysis of Agent Programs\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentFlow: Building Agent Dependency Graphs for Static Analysis of Agent Programs\"", + "authors": "Shenao Wang, Xinyi Hou, Yanjie Zhao, Xiao Cheng, Haoyu Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01640", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01641-when-agents-do-not-stop-uncovering-infinite-agentic-loops-in-llm-agents.md", + "title": "\"When Agents Do Not Stop: Uncovering Infinite Agentic Loops in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When Agents Do Not Stop: Uncovering Infinite Agentic Loops in LLM Agents\"", + "authors": "Xinyi Hou, Shenao Wang, Yanjie Zhao, Haoyu Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01641", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "llm-agent, planning-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01661-diverse-evidence-better-forecasts-multi-agent-deliberation-under-information-asy.md", + "title": "\"Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry\"", + "authors": "Yuante Li, Yicheng Tao, Kate Zhang, Taozhi Wang, Gefei Gu, Yaxin Zhou", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01661", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01668-verichat-an-agentic-conversational-ai-assistant-for-hardware-security-verificati.md", + "title": "\"VeriChat: An Agentic Conversational AI Assistant for Hardware Security Verification\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VeriChat: An Agentic Conversational AI Assistant for Hardware Security Verification\"", + "authors": "Dipayan Saha, Khan Thamid Hasan, Shams Tarek, Sujan Kumar Saha, Mark Tehranipoor, Farimah Farahmandi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01668", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01709-comfyclaw-self-evolving-skill-harnesses-for-image-generation-workflows.md", + "title": "\"COMFYCLAW: Self-Evolving Skill Harnesses for Image Generation Workflows\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"COMFYCLAW: Self-Evolving Skill Harnesses for Image Generation Workflows\"", + "authors": "Zongxia Li, Dawei Liu, Fuxiao Liu, Yuhang Zhou, Xiyang Wu, Jingxi Chen, Jing Xie, Xiaomin Wu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01709", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01766-simworlds-a-multi-agent-system-for-dynamic-3d-scene-creation.md", + "title": "\"SimWorlds: A Multi-Agent System for Dynamic 3D Scene Creation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SimWorlds: A Multi-Agent System for Dynamic 3D Scene Creation\"", + "authors": "Chunjiang Liu, Xiaoyuan Wang, Haoyu Chen, Yizhou Zhao, Ming-Hsuan Yang, László A. Jeni", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01766", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01767-repair-the-amplifier-not-the-symptom-stable-world-model-correction-for-agent-rol.md", + "title": "\"Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts\"", + "authors": "Xinyuan Song, Zekun Cai", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01767", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01788-krca-an-efficient-root-cause-analysis-system-in-hyper-scale-microservice-systems.md", + "title": "\"KRCA: An Efficient Root Cause Analysis System in Hyper-Scale Microservice Systems via Agentic AI\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"KRCA: An Efficient Root Cause Analysis System in Hyper-Scale Microservice Systems via Agentic AI\"", + "authors": "Jiamin Jiang, Jingfei Feng, Yu Luo, Qingliang Zhang, Yongqian Su, Wenwei Gu, Shenglin Zhang, Tianyu Cui, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01788", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01793-safety-testing-llm-agents-at-scale-from-risk-discovery-to-evidence-grounded-veri.md", + "title": "\"Safety Testing LLM Agents at Scale: From Risk Discovery to Evidence-Grounded Verification\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Safety Testing LLM Agents at Scale: From Risk Discovery to Evidence-Grounded Verification\"", + "authors": "Yunhao Feng, Ruixiao Lin, Ming Wen, Qinqin He, Yanming Guo, Yifan Ding, Yutao Wu, Jialuo Chen, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01793", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01812-to-master-an-llm-agent-framework-for-automated-topology-optimization.md", + "title": "\"TO-Master: an LLM-agent framework for automated topology optimization\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TO-Master: an LLM-agent framework for automated topology optimization\"", + "authors": "Haoju Lin, Wenchang Zhang, Weipeng Xu, Xiang Li, Tian Xu, Tianju Xue", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01812", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01874-skillcoach-self-evolving-rubrics-for-evaluating-and-enhancing-agentic-skill-use.md", + "title": "\"SkillCoach: Self-Evolving Rubrics for Evaluating and Enhancing Agentic Skill-Use\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SkillCoach: Self-Evolving Rubrics for Evaluating and Enhancing Agentic Skill-Use\"", + "authors": "Jiayin Zhu, Kelong Mao, Yudong Guo, Dengbo He, Sulong Xu, Simiu Gu, Yutao Yue", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01874", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01916-contextsniper-anttrail-s-token-efficient-code-memory-for-repository-level-progra.md", + "title": "\"ContextSniper: AntTrail's Token-Efficient Code Memory for Repository-Level Program Repair\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ContextSniper: AntTrail's Token-Efficient Code Memory for Repository-Level Program Repair\"", + "authors": "Chiwang Luk, Matin Mohammad Najafi, Zhifeng Jia, Wei Yang, Xiuchang Li, Jinwei Zhu, Yang Ren, Lei Chen, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01916", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory, coding-agent, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01929-beyond-textual-repository-exploration-dual-modal-structural-reasoning-for-agenti.md", + "title": "\"Beyond Textual Repository Exploration: Dual-Modal Structural Reasoning for Agentic Issue Resolution\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Textual Repository Exploration: Dual-Modal Structural Reasoning for Agentic Issue Resolution\"", + "authors": "Jiayi Zhang, Kai Huang, Yang Liu, Chunyang Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01929", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "coding-agent, function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-01935-a-tma-decoupling-state-aware-memory-failures-in-long-term-agent-memory.md", + "title": "\"A-TMA: Decoupling State-Aware Memory Failures in Long-Term Agent Memory\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"A-TMA: Decoupling State-Aware Memory Failures in Long-Term Agent Memory\"", + "authors": "Zitong Shi, Yixuan Tang, Anthony Kum Hoe Tung", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.01935", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-memory, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02032-pace-a-proxy-for-agentic-capability-evaluation.md", + "title": "\"PACE: A Proxy for Agentic Capability Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PACE: A Proxy for Agentic Capability Evaluation\"", + "authors": "Yueqi Song, Lintang Sutawika, Jiarui Liu, Lindia Tjuatja, Jiayi Geng, Yunze Xiao, Daniel Lee, Aditya Bharat Soni, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02032", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "agent-evaluation, coding-agent, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02186-ua-chatdev-uncertainty-aware-multi-agent-collaboration-for-reliable-software-dev.md", + "title": "\"UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development\"", + "authors": "Temitayo Olamilekan Ogunsusi, Lijun Qian, Xishuang Dong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02186", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02210-criticality-based-guard-rail-validation-for-ai-agent-decisions-in-autonomous-tel.md", + "title": "Criticality-Based Guard Rail Validation for AI Agent Decisions in Autonomous Telecom Networks", + "type": "paper", + "meta": { + "type": "paper", + "title": "Criticality-Based Guard Rail Validation for AI Agent Decisions in Autonomous Telecom Networks", + "authors": "Ravi Kant Sharma", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02210", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.NI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02245-copewell-a-multi-agent-swarm-architecture-for-equitable-mental-wellness-support.md", + "title": "\"Copewell: A Multi-Agent Swarm Architecture for Equitable Mental Wellness Support\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Copewell: A Multi-Agent Swarm Architecture for Equitable Mental Wellness Support\"", + "authors": "Seren Yenikent, Jack Vinijtrongjit, Katherine Ng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02245", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CY", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02255-agenticsts-a-bounded-memory-testbed-for-long-horizon-llm-agents.md", + "title": "\"AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents\"", + "authors": "Xiangchen Cheng, Yunwei Jiang, Jianwen Sun, Zizhen Li, Chuanhao Li, Xiangcheng Cao, Yihao Liu, Fanrui Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02255", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "23", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02294-coding-agents-are-guessing-measuring-action-boundary-violations-in-underspecifie.md", + "title": "\"Coding Agents Are Guessing: Measuring Action-Boundary Violations in Underspecified DevOps Instructions\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Coding Agents Are Guessing: Measuring Action-Boundary Violations in Underspecified DevOps Instructions\"", + "authors": "Zimo Ji, Zekai Zhang, Congying Xu, Zongjie Li, Yudong Gao, Shuai Wang, Shing-Chi Cheung", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02294", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02381-hulat2-at-mer-trans-2026-governed-multi-agent-simplification-for-spanish-easy-to.md", + "title": "\"HULAT2 at MER-TRANS 2026: Governed Multi-Agent Simplification for Spanish Easy-to-Read Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"HULAT2 at MER-TRANS 2026: Governed Multi-Agent Simplification for Spanish Easy-to-Read Generation\"", + "authors": "Lourdes Moreno, Paloma Martínez, Marco Antonio Sanchez-Escudero, Miguel Domínguez-Gómez", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02381", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02448-agentscad-automated-design-for-manufacturing-of-fdm-parts-via-multi-agent-llm-re.md", + "title": "\"AgentsCAD: Automated Design for Manufacturing of FDM Parts via Multi-Agent LLM Reasoning and Geometric Feature Recognition\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentsCAD: Automated Design for Manufacturing of FDM Parts via Multi-Agent LLM Reasoning and Geometric Feature Recognition\"", + "authors": "Emmanuel George, Christopher Keefe, Peter Pak, Amir Barati Farimani", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02448", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02453-adoption-and-ecosystem-health-a-longitudinal-analysis-of-open-source-multi-agent.md", + "title": "\"Adoption and Ecosystem Health: A Longitudinal Analysis of Open-Source Multi-Agent Frameworks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Adoption and Ecosystem Health: A Longitudinal Analysis of Open-Source Multi-Agent Frameworks\"", + "authors": "Xi Zhang, Papi Menon, Vivian Chu, Koray Cosguner", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02453", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02507-what-llm-agents-say-when-no-one-is-watching-social-structure-and-latent-objectiv.md", + "title": "\"What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates\"", + "authors": "Arman Ghaffarizadeh, Danyal Mohaddes, Aliakbar Izadkhah, Shahriar Noroozizadeh", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02507", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation, llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02577-benchmarking-the-benchmarks-a-validity-audit-of-tool-calling-evaluation.md", + "title": "\"Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation\"", + "authors": "Vishvesh Bhat, Jay Vaghasiya, Muhammad Ahmed Mohsin, Asad Aali", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02577", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02579-when-not-to-write-memory-governing-false-promotion-from-correlated-agent-traces.md", + "title": "\"When Not to Write Memory: Governing False Promotion from Correlated Agent Traces\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When Not to Write Memory: Governing False Promotion from Correlated Agent Traces\"", + "authors": "Yijiashun Qi, Xiang Xu, Yuxuan Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02579", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-06-30", + "updated_at": "2026-06-30", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-memory, language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02599-agentltl-a-trace-verification-framework-for-measuring-enforcing-and-training-pro.md", + "title": "\"AgentLTL: A Trace-Verification Framework for Measuring, Enforcing, and Training Procedural Compliance in Tool-Using LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentLTL: A Trace-Verification Framework for Measuring, Enforcing, and Training Procedural Compliance in Tool-Using LLM Agents\"", + "authors": "Laïla Elkoussy, Julien Perez", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02599", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.LO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02606-chainswe-benchmarking-coding-agents-on-multi-bug-software-maintenance.md", + "title": "\"ChainSWE: Benchmarking Coding Agents on Multi-Bug Software Maintenance\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ChainSWE: Benchmarking Coding Agents on Multi-Bug Software Maintenance\"", + "authors": "Qirui Jin, Lingching Tung, Kenan Li, Qiyang Shi, Yushi She, Huanzhong Jia, Harrison Zhao, Kejing Xia, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02606", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02684-can-coding-agents-implement-missed-compiler-optimizations-evaluating-llm-agents-.md", + "title": "Can Coding Agents Implement Missed Compiler Optimizations? Evaluating LLM Agents on LLVM Peephole Optimizations", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can Coding Agents Implement Missed Compiler Optimizations? Evaluating LLM Agents on LLVM Peephole Optimizations", + "authors": "Hongxu Xu, Chunhao Liao, Xintong Zhou, Chengnian Sun", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02684", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "coding-agent, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02689-s-ember-a-large-scale-benchmark-for-streaming-egocentric-memory-retrieval.md", + "title": "\"S-EMBER: A Large-Scale Benchmark for Streaming Egocentric Memory Retrieval\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"S-EMBER: A Large-Scale Benchmark for Streaming Egocentric Memory Retrieval\"", + "authors": "Xiaodong Wang, Xuanyi Zhao, Pedro Rodriguez, Devendra Singh Sachan, Barlas Oguz, Seungwhan Moon, Shang-Wen Li, Gargi Ghosh, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02689", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02703-llmoxie-exploring-agentic-ai-for-scientific-software-development.md", + "title": "\"LLMoxie: Exploring Agentic AI for Scientific Software Development\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LLMoxie: Exploring Agentic AI for Scientific Software Development\"", + "authors": "Landung Setiawan, Anant Mittal, Cordero Core, Anshul Tambay, Carlos Garcia Jurado Suarez, David A. C. Beck, Andrew J. Connolly, Vani Mandava", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02703", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "planning", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.DC", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agentic-ai, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02716-evaluating-large-language-models-for-decision-making-in-agent-based-urban-mobili.md", + "title": "Evaluating Large Language Models for Decision-Making in Agent-Based Urban Mobility Simulations", + "type": "paper", + "meta": { + "type": "paper", + "title": "Evaluating Large Language Models for Decision-Making in Agent-Based Urban Mobility Simulations", + "authors": "Bruno Cascaes Alves, Míriam Blank Born, Ulisses Gilioli Francescatto Júnior, Felipe Moura Goulart, Letícia Brandão Caldas, Marilton Sanchotene de Aguiar", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02716", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "multi-agent", + "planning", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02807-swarmresearch-orchestrating-coding-agents-for-open-ended-discovery.md", + "title": "\"SwarmResearch: Orchestrating Coding Agents for Open-Ended Discovery\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SwarmResearch: Orchestrating Coding Agents for Open-Ended Discovery\"", + "authors": "Yuvraj Virk, Zack Edds, Chunqiu Steven Xia, Lingming Zhang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02807", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-02", + "updated_at": "2026-07-02", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "computer-use", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02846-object-centric-environment-modeling-for-agentic-tasks.md", + "title": "Object-Centric Environment Modeling for Agentic Tasks", + "type": "paper", + "meta": { + "type": "paper", + "title": "Object-Centric Environment Modeling for Agentic Tasks", + "authors": "Yiyang Li, Tianyi Ma, Zehong Wang, Yijun Ma, Yanfang Ye", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02846", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02857-mosaic-knowledge-guided-cli-command-composition-attack-in-llm-coding-agents.md", + "title": "\"MOSAIC: Knowledge-Guided CLI Command Composition Attack in LLM Coding Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MOSAIC: Knowledge-Guided CLI Command Composition Attack in LLM Coding Agents\"", + "authors": "Jiangrong Wu, Huaijin Wang, Yihao Zhang, Yuhong Nan, Shuai Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02857", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02879-medcalc-pro-solving-complex-medical-calculations-with-llm-agents.md", + "title": "\"MedCalc-Pro: Solving Complex Medical Calculations with LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MedCalc-Pro: Solving Complex Medical Calculations with LLM Agents\"", + "authors": "Siran Zhao, Ruihui Hou, Ziyue Huai, Chennuo Zhang, Tong Ruan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02879", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02882-diagnosis-driven-automatic-repair-for-agentic-workflow-via-symbolic-inference.md", + "title": "Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference", + "type": "paper", + "meta": { + "type": "paper", + "title": "Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference", + "authors": "Xuyan Ma, Yawen Wang, Junjie Wang, Xiaofei Xie, Boyu Wu, Mingyang Li, Dandan Wang, Qing Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02882", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02911-coact-action-preserving-observation-compression-for-coding-agents.md", + "title": "\"CoACT: Action-Preserving Observation Compression for Coding Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CoACT: Action-Preserving Observation Compression for Coding Agents\"", + "authors": "Haorui Chen, Yuancheng Zhu, Yitong Zhang, Jia Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02911", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02927-videosearcher-empowering-video-deep-research-with-multi-tool-agentic-reasoning-v.md", + "title": "\"VideoSearcher: Empowering Video Deep Research with Multi-Tool Agentic Reasoning via Reinforcement Learning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"VideoSearcher: Empowering Video Deep Research with Multi-Tool Agentic Reasoning via Reinforcement Learning\"", + "authors": "Zhenkun Gao, Yicheng Bao, Jinlong Peng, Xueheng Li, Theo Huang, Bangwei Liu, Kunquan Li, Zhenye Gan, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02927", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-02942-a-workflow-aware-serving-layer-for-agentic-applications.md", + "title": "A Workflow-Aware Serving Layer for Agentic Applications", + "type": "paper", + "meta": { + "type": "paper", + "title": "A Workflow-Aware Serving Layer for Agentic Applications", + "authors": "Jiayi Qian, Zishen Wan, Hanchen Yang, Chun Tao, Souvik Kundu, Tushar Krishna", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.02942", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DC", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03105-orbit-q-dual-axis-benchmarking-of-autonomous-agents-in-scientific-quantum-progra.md", + "title": "\"ORBIT-Q: Dual-axis benchmarking of autonomous agents in scientific quantum programming\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ORBIT-Q: Dual-axis benchmarking of autonomous agents in scientific quantum programming\"", + "authors": "Shi-Xin Zhang, Yu-Qin Chen", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03105", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "quant-ph" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03162-apeb-benchmarking-personalization-ability-of-large-language-model-agents.md", + "title": "\"APeB: Benchmarking Personalization Ability of Large Language Model Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"APeB: Benchmarking Personalization Ability of Large Language Model Agents\"", + "authors": "Garry Yang, Zizhe Chen, Xinru Chen, Yongqiang Chen, Jianxiang Wang, Deyu Zou, Linyi Ding, Jialiang Wu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03162", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03220-contra-red-teaming-configurations-of-personalizable-agents.md", + "title": "\"CONTRA: Red-Teaming Configurations of Personalizable Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CONTRA: Red-Teaming Configurations of Personalizable Agents\"", + "authors": "Jonathan Nöther, Adish Singla, Goran Radanovic", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03220", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03233-agentic-and-generative-ai-for-open-source-intelligence-and-cyber-investigations-.md", + "title": "\"Agentic and Generative AI for Open-Source Intelligence and Cyber Investigations: Taxonomy, Evaluation, Challenges, and Future Directions\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic and Generative AI for Open-Source Intelligence and Cyber Investigations: Taxonomy, Evaluation, Challenges, and Future Directions\"", + "authors": "Eduardo Almeida Palmieri, Mohamed Chahine Ghanem, Dipo Dunsin, Zubair Baig, Ed de Quincey, Kim-Kwang Raymond Choo", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03233", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.IR", + "cs.SI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "22", + "collection_queries": "agentic-ai, rag-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03269-agentic-secpbft-agentic-ai-driven-proactive-security-framework-for-wireless-pbft.md", + "title": "\"Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks\"", + "authors": "Haoxiang Luo, Yinqiu Liu, Ruichen Zhang, Guangyuan Liu, Gang Sun, Hongfang Yu, Zhu Han, Dong In Kim", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03269", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "computer-use", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.NI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03316-is-agentic-code-review-helpful-mining-developers-feedback-to-coderabbit-reviews-.md", + "title": "\"Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild\"", + "authors": "Hong Yi Lin, Mingzhao Liang, Kla Tantithamthavorn, Patanamon Thongtanunam", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03316", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "autonomous-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03333-spork-self-speculative-forking-to-accelerate-agentic-llm-inference.md", + "title": "\"SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference\"", + "authors": "Huajun Bai, Weiwei Lv, Huichuan Zheng, Youyou Lu, Jiwu Shu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03333", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DC", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03423-securing-multi-tool-ai-agent-chains-with-dynamic-real-time-compositional-policie.md", + "title": "Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies", + "type": "paper", + "meta": { + "type": "paper", + "title": "Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies", + "authors": "Chris Schneider, Kriti Faujdar, Philipp Schoenegger, Ben Bariach", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03423", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "ai-agent, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03441-no-time-like-the-present-agentic-test-time-training-for-llm-agents.md", + "title": "\"No Time Like the Present: Agentic Test-Time Training for LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"No Time Like the Present: Agentic Test-Time Training for LLM Agents\"", + "authors": "Yanbo Wang, Jinhua Hao, Yuze Shi, Kun Yuan, Ming Sun", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03441", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "coding-agent, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03510-cage-1-control-assurance-and-governance-evaluation-for-enterprise-agentic-ai.md", + "title": "\"CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI\"", + "authors": "Roopam W. Sure", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03510", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03525-gameenginebench-evaluating-coding-agents-on-real-c-runtime-environments.md", + "title": "\"GameEngineBench: Evaluating Coding Agents on Real C++ Runtime Environments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"GameEngineBench: Evaluating Coding Agents on Real C++ Runtime Environments\"", + "authors": "Brian La, Sejoon Chang, Ben Kim, Junyoung Bae, Aamish Ahmad Beg, Sei Chang, Gonzalo Gonzalez-Pumariega", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03525", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "memory", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03601-archeval-measuring-ai-agents-as-computer-architects.md", + "title": "\"ArchEval: Measuring AI Agents as Computer Architects\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ArchEval: Measuring AI Agents as Computer Architects\"", + "authors": "Chenyu Wang, Zishen Wan, Jeffrey Ma, Shvetank Prakash, Zhenting Qi, Haebin Do, Andy Cheng, Arya Tschand, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03601", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "ai-agent, llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03628-swarm-driven-multi-agent-reasoning-for-smart-city-security.md", + "title": "Swarm-Driven Multi-Agent Reasoning for Smart City Security", + "type": "paper", + "meta": { + "type": "paper", + "title": "Swarm-Driven Multi-Agent Reasoning for Smart City Security", + "authors": "Saeid Jamshidi, Kawser Wazed Nafi, Carol Fung, Foutse Khomh", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03628", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-03", + "updated_at": "2026-07-03", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03691-don-t-blame-the-large-language-model-how-scaffolding-evolution-shapes-coding-age.md", + "title": "\"Don't Blame the Large Language Model: How Scaffolding Evolution Shapes Coding Agent Quality\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Don't Blame the Large Language Model: How Scaffolding Evolution Shapes Coding Agent Quality\"", + "authors": "Oussama Ben Sghaier, Hao Li, Bram Adams, Ahmed E. Hassan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03691", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03695-social-networks-of-llm-agents.md", + "title": "Social Networks of LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Social Networks of LLM Agents", + "authors": "Kaixuan Liu, Guojun Xiong, Weinan Zhang, Shengpu Tang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03695", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03702-agent-reinforcement-learning-via-pivotal-aware-self-feedback-retry.md", + "title": "Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry", + "authors": "Weiyang Guo, Zesheng Shi, Longhui Zhang, Zeen Zhu, Min Zhang, Jing Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03702", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03726-selfmem-self-optimizing-memory-for-ai-agents.md", + "title": "\"SelfMem: Self-Optimizing Memory for AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SelfMem: Self-Optimizing Memory for AI Agents\"", + "authors": "Shu Yang, Junchao Wu, Derek F. Wong, Di Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03726", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-memory, ai-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03821-dualview-preventing-indirect-prompt-injection-in-personal-ai-agents.md", + "title": "\"DualView: Preventing Indirect Prompt Injection in Personal AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"DualView: Preventing Indirect Prompt Injection in Personal AI Agents\"", + "authors": "Juhee Kim, Woohyuk Choi, Taehyun Kang, Youngmin Kim, Byoungyoung Lee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03821", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03853-cograd-a-cognitively-inspired-multi-agent-framework-for-radiology-report-generat.md", + "title": "\"CogRad: A Cognitively-Inspired Multi-Agent Framework for Radiology Report Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CogRad: A Cognitively-Inspired Multi-Agent Framework for Radiology Report Generation\"", + "authors": "Saif Ur Rehman Khan, Hasaan Maqsood, Sebastian Vollmer, Andreas Dengel, Muhammad Nabeel Asim", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03853", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03953-the-remarkable-effectiveness-of-providing-ai-agents-with-natural-language-tools-.md", + "title": "\"The Remarkable Effectiveness of Providing AI Agents with Natural Language Tools: A Replication Study Validating NLT Performance Across 14 Models\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Remarkable Effectiveness of Providing AI Agents with Natural Language Tools: A Replication Study Validating NLT Performance Across 14 Models\"", + "authors": "Alexander Somma, Isabelle Plante, Fred Premji", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03953", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agentic-ai, ai-agent, llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-03968-refused-in-chat-written-in-code-workflow-level-jailbreak-construction-in-ide-cod.md", + "title": "\"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents\"", + "authors": "Abhishek Kumar, Carsten Maple", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.03968", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04009-physminer-an-agentic-ai-framework-for-discovering-turbulence-physics.md", + "title": "\"PhysMiner: An Agentic AI Framework for Discovering Turbulence Physics\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PhysMiner: An Agentic AI Framework for Discovering Turbulence Physics\"", + "authors": "Jiawei Chen, Han Gao, Ping He", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04009", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "physics.flu-dyn" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04034-the-i-don-t-know-filter-enhancing-agentic-reliability-in-function-calling.md", + "title": "\"The \\\"I Don't Know\\\" Filter: Enhancing Agentic Reliability in Function Calling\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The \\\"I Don't Know\\\" Filter: Enhancing Agentic Reliability in Function Calling\"", + "authors": "Stefan Broecker, Mason del Rosario, Boris Selitser, Thomas Strohmer", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04034", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-04", + "updated_at": "2026-07-04", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation, function-calling" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04089-placemem-toward-a-compute-aware-memory-plane-for-lifelong-agents.md", + "title": "\"PLACEMEM: Toward a Compute-Aware Memory Plane for Lifelong Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PLACEMEM: Toward a Compute-Aware Memory Plane for Lifelong Agents\"", + "authors": "Sukanta Ganguly", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04089", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agent-memory" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04149-beyond-scene-priors-fine-grained-traffic-scene-reasoning-with-benchmarking-and-q.md", + "title": "\"Beyond Scene Priors: Fine-Grained Traffic Scene Reasoning with Benchmarking and Query-Guided Small-Object Focus\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Scene Priors: Fine-Grained Traffic Scene Reasoning with Benchmarking and Query-Guided Small-Object Focus\"", + "authors": "Waikit Xiu, Qiang Lu, Zian Wang, Xinjie Yang, Zhiwei Chen, Chen Sun, Xiying Li", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04149", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04162-ace-agentic-control-for-embodied-manipulation-via-zero-shot-workflow-reasoning.md", + "title": "\"ACE: Agentic Control for Embodied Manipulation via Zero-shot Workflow Reasoning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ACE: Agentic Control for Embodied Manipulation via Zero-shot Workflow Reasoning\"", + "authors": "Iok Tong Lei, QianZhi Li, Ying Jie Yap, Yujie Zhang, Rui Zhong, Haichao Gui, Xiaolong Liu, Zhidong Deng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04162", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO", + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04212-an-evaluation-of-role-based-multi-agent-code-generation-on-repository-scale-prob.md", + "title": "An Evaluation of Role-Based Multi-Agent Code Generation on Repository-Scale Problems", + "type": "paper", + "meta": { + "type": "paper", + "title": "An Evaluation of Role-Based Multi-Agent Code Generation on Repository-Scale Problems", + "authors": "Benedetta Donato, Noah Hagar-Dent, Aaron Worsnop, Leonardo Mariani, Valerio Terragni", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04212", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04219-agentic-iot-architectures-applications-and-challenges-toward-the-internet-of-age.md", + "title": "\"Agentic IoT: Architectures, Applications, and Challenges Toward the Internet of Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agentic IoT: Architectures, Applications, and Challenges Toward the Internet of Agents\"", + "authors": "Rümeysa Hilal Sevinç, Bahaeddin Türkoğlu, İbrahim Kök", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04219", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA", + "cs.NI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "ai-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04240-biological-motifs-for-agentic-control.md", + "title": "Biological Motifs for Agentic Control", + "type": "paper", + "meta": { + "type": "paper", + "title": "Biological Motifs for Agentic Control", + "authors": "Bogdan Banu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04240", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "multi-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "q-bio.CB" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agent-evaluation, autonomous-agent-llm, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04293-causalgame-benchmarking-causal-thinking-of-llm-agents-in-games.md", + "title": "\"CausalGame: Benchmarking Causal Thinking of LLM Agents in Games\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CausalGame: Benchmarking Causal Thinking of LLM Agents in Games\"", + "authors": "Zhenhao Chen, Yongqiang Chen, Chenxi Liu, Junchi Yu, Xiangchen Song, Zijian Li, Jialin Li, Philip Torr, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04293", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.LG", + "stat.ML" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04334-do-gui-agents-believe-their-eyes-diagnosing-state-belief-reliance-on-pixels-vers.md", + "title": "Do GUI Agents Believe Their Eyes? Diagnosing State-Belief Reliance on Pixels versus Structure", + "type": "paper", + "meta": { + "type": "paper", + "title": "Do GUI Agents Believe Their Eyes? Diagnosing State-Belief Reliance on Pixels versus Structure", + "authors": "Guijia Zhang, Harry Yang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04334", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04391-memory-orchestrated-semantic-system-moss-an-auditable-agentic-memory-architectur.md", + "title": "\"Memory-Orchestrated Semantic System (MOSS): An Auditable Agentic Memory Architecture\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Memory-Orchestrated Semantic System (MOSS): An Auditable Agentic Memory Architecture\"", + "authors": "Serge Lacasse, Jérémie Hatier, Alex Baker", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04391", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "agent-memory, ai-agent, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04394-mechmath-agent-team-llm-driven-agents-for-mathematical-research.md", + "title": "\"MechMath Agent Team: LLM Driven Agents for Mathematical Research\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MechMath Agent Team: LLM Driven Agents for Mathematical Research\"", + "authors": "Yichuan Cao, Ruichen Qiu, Junqi Liu, Jiaqi Wang, Dakai Guo, Ruyong Feng, Lihong Zhi, Xiao-Shan Gao", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04394", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "planning", + "rag", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.SC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04395-nki-agent-domain-specific-fine-tuning-and-agentic-tool-use-for-neuron-kernel-gen.md", + "title": "\"NKI-Agent: Domain-Specific Fine-Tuning and Agentic Tool Use for Neuron Kernel Generation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"NKI-Agent: Domain-Specific Fine-Tuning and Agentic Tool Use for Neuron Kernel Generation\"", + "authors": "Junjie Tang, Jun Huan, Hao Zhou, Yuhao Zhang, Lin Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04395", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04426-ace-brain-0-5-a-unified-embodied-foundational-model-for-physical-agentic-ai.md", + "title": "\"ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI\"", + "authors": "\"ACE-Brain Team, :, Ziyang Gong, Haoming Gu, Zehang Luo, Tianyi Zhang, Tao Tao, Yixiao Chi, et al.\"", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04426", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agentic-ai" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04433-autonomous-information-seeking-a-roadmap-for-agentic-recommender-systems.md", + "title": "\"Autonomous Information Seeking: A Roadmap for Agentic Recommender Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Autonomous Information Seeking: A Roadmap for Agentic Recommender Systems\"", + "authors": "Xinyu Lin, Yashar Deldjoo, Sunhao Dai, Honghui Bao, Xiaopeng Ye, Fatemeh Nazary, Wenjie Wang, Tommaso Di Noia, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04433", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "planning", + "reasoning", + "tool-use", + "workflow-agent", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.IR", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04470-regime-conditional-stabilisation-of-llm-augmented-cooperative-multi-agent-reinfo.md", + "title": "Regime-Conditional Stabilisation of LLM-Augmented Cooperative Multi-Agent Reinforcement Learning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Regime-Conditional Stabilisation of LLM-Augmented Cooperative Multi-Agent Reinforcement Learning", + "authors": "Faid Keddouri, Sohaib Houhou, Aissa Boulmerka, Nadir Farhi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04470", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI", + "math.OC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04528-measuring-harness-induced-belief-divergence-in-multi-step-llm-agents.md", + "title": "Measuring Harness-Induced Belief Divergence in Multi-Step LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Measuring Harness-Induced Belief Divergence in Multi-Step LLM Agents", + "authors": "Haiwen Yi, Xinyuan Song", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04528", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04569-llms-for-agentic-home-energy-management.md", + "title": "LLMs for Agentic Home Energy Management", + "type": "paper", + "meta": { + "type": "paper", + "title": "LLMs for Agentic Home Energy Management", + "authors": "Sokipriala Jonah", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04569", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "eess.SY" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "function-calling, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04617-mrms-a-multi-resolution-memory-substrate-for-long-lived-ai-agents.md", + "title": "\"MRMS: A Multi-Resolution Memory Substrate for Long-Lived AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MRMS: A Multi-Resolution Memory Substrate for Long-Lived AI Agents\"", + "authors": "Jizhizi Li, Amy Shi-Nash", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04617", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04623-can-llms-really-recover-microservice-failures-a-recovery-aware-evaluation-of-dia.md", + "title": "Can LLMs Really Recover Microservice Failures? A Recovery-Aware Evaluation of Diagnosis-to-Action Reasoning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Can LLMs Really Recover Microservice Failures? A Recovery-Aware Evaluation of Diagnosis-to-Action Reasoning", + "authors": "Jiaxing Qi, Zhongzhi Luan, Hongyu Zhang, Shaohan Huang, Carol Fung, Yongxin Tong, Hailong Yang, Depei Qian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04623", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.DC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04686-toolfailbench-diagnosing-tool-use-failures-in-llm-agents.md", + "title": "\"ToolFailBench: Diagnosing Tool-Use Failures in LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"ToolFailBench: Diagnosing Tool-Use Failures in LLM Agents\"", + "authors": "Harsh Soni", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04686", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "agentic-ai, llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04697-ai-agent-pull-requests-on-github-frequency-structure-and-merge-conflict-rates.md", + "title": "\"AI Agent Pull Requests on GitHub: Frequency, Structure, and Merge Conflict Rates\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AI Agent Pull Requests on GitHub: Frequency, Structure, and Merge Conflict Rates\"", + "authors": "George Xu, Arjun Subramanian, Nithilan Karthik", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04697", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04713-rspo-reward-swap-policy-optimization-for-multi-turn-llm-agents.md", + "title": "\"RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents\"", + "authors": "Qiang Liu, Taian Guo, Ruizhi Qiao, Xing Sun", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04713", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-04963-stapo-selective-trajectory-aware-policy-optimization-for-llm-agent-training.md", + "title": "\"STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training\"", + "authors": "Qiuyi Qi, Tian Liang, Mutian Bao, Jinjian Zhang, Dongnan Liu, Wei Zhou, Linjian Mo, Ming Kong, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.04963", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "planning", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05001-tactic-kg-toward-small-agent-teams-for-cyber-threat-intelligence-knowledge-graph.md", + "title": "\"TACTIC-KG: Toward Small Agent Teams for Cyber Threat Intelligence Knowledge Graph Construction\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"TACTIC-KG: Toward Small Agent Teams for Cyber Threat Intelligence Knowledge Graph Construction\"", + "authors": "Mouhamed Amine Bouchiha, Gregory Blanc", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05001", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI", + "cs.LG", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05029-your-agent-s-memories-are-not-its-own-forged-reasoning-attacks-on-llm-agent-memo.md", + "title": "\"Your Agent's Memories Are Not Its Own: Forged Reasoning Attacks on LLM Agent Memory and Defenses\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Your Agent's Memories Are Not Its Own: Forged Reasoning Attacks on LLM Agent Memory and Defenses\"", + "authors": "Neeraj Karamchandani, Piyush Nagasubramaniam, Sencun Zhu, Dinghao Wu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05029", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-memory, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05055-toward-trustworthy-large-language-model-agents-in-healthcare.md", + "title": "Toward Trustworthy Large Language Model Agents in Healthcare", + "type": "paper", + "meta": { + "type": "paper", + "title": "Toward Trustworthy Large Language Model Agents in Healthcare", + "authors": "Hadi Hasan, Safaa Salman, Adam Tai Abou Dargham, Ammar Mohanna, Ali Chehab", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05055", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "function-calling, rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05120-agent-data-injection-attacks-are-realistic-threats-to-ai-agents.md", + "title": "Agent Data Injection Attacks are Realistic Threats to AI Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agent Data Injection Attacks are Realistic Threats to AI Agents", + "authors": "Woohyuk Choi, Juhee Kim, Taehyun Kang, Jihyeon Jeong, Luyi Xing, Byoungyoung Lee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05120", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "computer-use", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-safety, ai-agent, coding-agent, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05132-when-agents-lie-premeditation-persistence-and-exploitation-in-repeated-games.md", + "title": "\"When Agents Lie: Premeditation, Persistence, and Exploitation in Repeated Games\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"When Agents Lie: Premeditation, Persistence, and Exploitation in Repeated Games\"", + "authors": "Jerick Shi, Terry Jingcheng Zhang, Bernhard Schölkopf, Vincent Conitzer, Zhijing Jin", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05132", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CY", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "autonomous-agent-llm, llm-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05174-agentgym2-benchmarking-large-language-model-agents-in-de-idealized-real-world-en.md", + "title": "\"AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments\"", + "authors": "Zhiheng Xi, Dingwen Yang, Jiaqi Liu, Jixuan Huang, Honglin Guo, Baodai Huang, Tinggang Chen, Qi Zhang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05174", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "language-agent, llm-agent, planning-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05188-latent-programming-horizons-in-coding-agents.md", + "title": "Latent Programming Horizons in Coding Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Latent Programming Horizons in Coding Agents", + "authors": "André Silva, Han Tu, Martin Monperrus", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05188", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05202-evoagentbench-benchmarking-agent-self-evolution-via-ability-transfer.md", + "title": "\"EvoAgentBench: Benchmarking Agent Self-Evolution via Ability Transfer\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"EvoAgentBench: Benchmarking Agent Self-Evolution via Ability Transfer\"", + "authors": "Xingze Gao, Chuanrui Hu, Hongda Chen, Pengfei Yao, Zhao Wang, Yi Bai, Zhengwei Wu, Yunyun Han, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05202", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "memory", + "planning", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "19", + "collection_queries": "agent-evaluation" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05297-metaskill-evolve-recursive-self-improvement-of-llm-agents-via-two-timescale-meta.md", + "title": "\"MetaSkill-Evolve: Recursive Self-Improvement of LLM Agents via Two-Timescale Meta-Skill Evolution\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"MetaSkill-Evolve: Recursive Self-Improvement of LLM Agents via Two-Timescale Meta-Skill Evolution\"", + "authors": "Zefeng Wang, Minxi Yan, Jinhe Bi, Sikuan Yan, Volker Tresp, Yunpu Ma", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05297", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "planning", + "reasoning", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "agent-evaluation, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05318-pisas-benchmarking-contextual-integrity-in-multi-user-agentic-systems.md", + "title": "\"PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems\"", + "authors": "Shubham Gupta, Nazanin Mohammadi Sepahvand, Abhinav Kumar, Cem Subakan, Spandana Gella, Pierre-André Noël, Perouz Taslakian, Eugene Bagdasarian, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05318", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.MA", + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "20", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05363-sovereignpa-bench-evaluating-user-owned-personal-agents-under-evolving-intent-pl.md", + "title": "\"SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints\"", + "authors": "Dylan Zongmin Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05363", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "memory", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05378-compactionrl-reinforcement-learning-with-context-compaction-for-long-horizon-age.md", + "title": "\"CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents\"", + "authors": "Yujiang Li, Zhenyu Hou, Yi Jing, Jie Tang, Yuxiao Dong", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05378", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent", + "planning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "coding-agent, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05391-llm-as-a-verifier-a-general-purpose-verification-framework.md", + "title": "\"LLM-as-a-Verifier: A General-Purpose Verification Framework\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LLM-as-a-Verifier: A General-Purpose Verification Framework\"", + "authors": "Jacky Kwok, Shulu Li, Pranav Atreya, Yuejiang Liu, Yixing Jiang, Chelsea Finn, Marco Pavone, Ion Stoica, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05391", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "embodied-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "cs.LG", + "cs.MA", + "cs.RO" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05428-charlie-an-on-premise-multi-agent-retrieval-augmented-generation-system-for-evid.md", + "title": "\"CHARLIE: An On-Premise Multi-Agent Retrieval-Augmented Generation System for Evidential Reasoning in Forensic Science\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CHARLIE: An On-Premise Multi-Agent Retrieval-Augmented Generation System for Evidential Reasoning in Forensic Science\"", + "authors": "Leandro D. Carneiro, Andre L. S. Meirelles, Juliano de A. Gomes, Rafael C. A. Cabral", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05428", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-01", + "updated_at": "2026-07-01", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "multi-agent", + "planning", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.DL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "rag-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05456-prompt-to-paper-agentic-ai-system-for-bioinformatics.md", + "title": "\"Prompt-to-Paper: Agentic AI System for Bioinformatics\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Prompt-to-Paper: Agentic AI System for Bioinformatics\"", + "authors": "Ramsha Kamran, Maheera Amjad, Zartasha Mustansar, Arsalan Shaukat, Salma Sherbaz, Muhammad U. S. Khan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05456", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL", + "q-bio.QM" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "agentic-ai, coding-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05458-learning-to-control-llm-agent-harnesses-with-offline-reinforcement-learning.md", + "title": "Learning to Control LLM Agent Harnesses with Offline Reinforcement Learning", + "type": "paper", + "meta": { + "type": "paper", + "title": "Learning to Control LLM Agent Harnesses with Offline Reinforcement Learning", + "authors": "Haiwen Yi, Xinyuan Song", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05458", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-05", + "updated_at": "2026-07-05", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.LG", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05518-aiauthz-off-host-identity-bound-authorization-for-ai-agents.md", + "title": "\"aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents\"", + "authors": "Sai Varun Kodathala", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05518", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05659-agents-with-feelings-personality-and-emotion-in-multi-agent-software-teams.md", + "title": "Agents with Feelings? Personality and Emotion in Multi-Agent Software Teams", + "type": "paper", + "meta": { + "type": "paper", + "title": "Agents with Feelings? Personality and Emotion in Multi-Agent Software Teams", + "authors": "Yunyan Ding, Thomas Zimmermann, Iftekhar Ahmed", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05659", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "multi-agent", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05666-what-do-ai-agents-actually-change-an-empirical-taxonomy-of-mutation-patterns-in-.md", + "title": "What Do AI Agents Actually Change? An Empirical Taxonomy of Mutation Patterns in Performance-Improving Pull Requests", + "type": "paper", + "meta": { + "type": "paper", + "title": "What Do AI Agents Actually Change? An Empirical Taxonomy of Mutation Patterns in Performance-Improving Pull Requests", + "authors": "Illia Dovhoshliubnyi, Nima Soroush, Ashkan Sami, Alexander Brownlee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05666", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "coding-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "ai-agent, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05677-from-conversation-to-contribution-characterizing-coding-agent-in-open-source-sof.md", + "title": "\"From Conversation to Contribution: Characterizing Coding Agent in Open-Source Software\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Conversation to Contribution: Characterizing Coding Agent in Open-Source Software\"", + "authors": "Zihan Fang, Yueke Zhang, Ningzhi Tang, Collin McMillan, Toby Jia-Jun Li, Yu Huang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05677", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "computer-use", + "multi-agent", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05690-memory-in-the-loop-in-process-retrieval-as-extendedworking-memory-for-language-a.md", + "title": "\"Memory in the Loop: In-Process Retrieval as ExtendedWorking Memory for Language Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Memory in the Loop: In-Process Retrieval as ExtendedWorking Memory for Language Agents\"", + "authors": "Yusuf Khan, Carlo Lipizzi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05690", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-06", + "updated_at": "2026-07-06", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "language-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05743-the-balkanization-of-execution-security-research-for-ai-coding-agents-isolation-.md", + "title": "\"The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities\"", + "authors": "Mohammadreza Rashidi", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05743", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05772-detecting-vulnerability-inducing-commits-via-multi-stage-reasoning-with-llm-base.md", + "title": "Detecting Vulnerability-Inducing Commits via Multi-Stage Reasoning with LLM-Based Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Detecting Vulnerability-Inducing Commits via Multi-Stage Reasoning with LLM-Based Agents", + "authors": "Liyou Chen, Hailong Sun, Xiang Gao, Yue Pan", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05772", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05773-beyond-static-evaluation-building-simulation-environments-for-scalable-agentic-r.md", + "title": "\"Beyond Static Evaluation: Building Simulation Environments for Scalable Agentic Reinforcement Learning\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond Static Evaluation: Building Simulation Environments for Scalable Agentic Reinforcement Learning\"", + "authors": "Akshay Arora, Ishan Nigam, Ashutosh Aggarwal, Shefali Bansal, Krishna Singh, Sweta Kumari, Nikhil Mittal, Shariq Farhan, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05773", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "autonomous-agent-llm, tool-use, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05775-beyond-the-leaderboard-a-synthesis-of-tool-use-planning-and-reasoning-failures-i.md", + "title": "\"Beyond the Leaderboard: A Synthesis of Tool-Use, Planning, and Reasoning Failures in Large Language Model Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Beyond the Leaderboard: A Synthesis of Tool-Use, Planning, and Reasoning Failures in Large Language Model Agents\"", + "authors": "Wael Albayaydh, Rui Zhao, Ivan Flechais", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05775", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "coding-agent", + "embodied-agent", + "multi-agent", + "planning", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "23", + "collection_queries": "llm-agent, multi-agent-llm, planning-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05794-from-passive-retrieval-to-active-memory-navigation-learning-to-use-memory-as-a-s.md", + "title": "\"From Passive Retrieval to Active Memory Navigation: Learning to Use Memory as a Structured Action Space\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Passive Retrieval to Active Memory Navigation: Learning to Use Memory as a Structured Action Space\"", + "authors": "Yue Xu, Yutao Sun, Yihao Liu, Mengyu Zhou, Jiayi Qiao, Lu Ma, Kai Tang, Wenjie Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05794", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "embodied-agent", + "memory", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05805-onnes-a-physics-grounded-multi-agent-llm-simulator-for-cryogenic-fault-diagnosis.md", + "title": "\"Onnes: A Physics-Grounded Multi-Agent LLM Simulator for Cryogenic Fault Diagnosis in Quantum Computing Infrastructure\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Onnes: A Physics-Grounded Multi-Agent LLM Simulator for Cryogenic Fault Diagnosis in Quantum Computing Infrastructure\"", + "authors": "Praneeth Narisetty, Uday Kumar Reddy Kattamanchi, Shiva Nagendra Babu Kore", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05805", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.LG", + "quant-ph" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-05915-pcbworld-a-benchmark-environment-for-engine-grounded-pcb-design-automation.md", + "title": "\"PCBWorld: A Benchmark Environment for Engine-Grounded PCB Design Automation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PCBWorld: A Benchmark Environment for Engine-Grounded PCB Design Automation\"", + "authors": "Hyungseok Song, Junseok Park, Won-Seok Choi, Seohui Bae, Han-Seul Jeong, Youngjoon Park, Soonyoung Lee", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.05915", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "agentic-ai, llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06000-context-to-execution-integrity-for-llm-agents.md", + "title": "Context-to-Execution Integrity for LLM Agents", + "type": "paper", + "meta": { + "type": "paper", + "title": "Context-to-Execution Integrity for LLM Agents", + "authors": "Igor Santos-Grueiro", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06000", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CR" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation, coding-agent, llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06001-information-limits-and-attractor-dynamics-in-economies-of-frontier-llm-agents-a-.md", + "title": "\"Information Limits and Attractor Dynamics in Economies of Frontier LLM Agents: A Pre-Registered Test\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Information Limits and Attractor Dynamics in Economies of Frontier LLM Agents: A Pre-Registered Test\"", + "authors": "Cheng Qian", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06001", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.MA" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06008-polyworkbench-benchmarking-multilingual-long-horizon-llm-agents.md", + "title": "\"PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents\"", + "authors": "Hongliang Li, Yijin Liu, Zhiwei Zhang, Zihe Liu, Xinyue Lou, Jinan Xu, Fandong Meng, Kaiyu Huang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06008", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "25", + "collection_queries": "agent-evaluation, llm-agent, planning-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06080-from-blueprint-to-reality-modeling-and-applying-putnam-s-social-capital-theory-w.md", + "title": "\"From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations\"", + "authors": "Shiyi Ling, Zhi Zheng, Hui Zheng, Wenjun Xue, Feng Ye, Tong Xu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06080", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "multi-agent", + "rag", + "tool-use", + "world-model" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI", + "cs.SI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06101-agents-that-teach-towards-designing-incidental-learning-back-into-ai-assisted-so.md", + "title": "\"Agents That Teach: Towards Designing Incidental Learning Back into AI-Assisted Software Development\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Agents That Teach: Towards Designing Incidental Learning Back into AI-Assisted Software Development\"", + "authors": "Rohit Mehra, Samdyuti Suri, Prithviraj K Tagadinamani, Kapil Singi, Vikrant Kaulgud, Adam P. Burden", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06101", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-safety", + "coding-agent", + "computer-use", + "multi-agent", + "rag", + "reasoning", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CY", + "cs.HC" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06118-webretriever-a-large-scale-comprehensive-benchmark-for-efficient-web-agent-evalu.md", + "title": "\"WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation\"", + "authors": "Wei Dong, Tianyu Fu, Zhe Yu, Hanning Wang, Anyang Su, Zhizhou Fang, Yuyang Chen, Shuo Wang, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06118", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "agent-safety", + "computer-use", + "embodied-agent", + "rag", + "tool-use", + "workflow-agent" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CV", + "cs.MM" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "21", + "collection_queries": "agent-evaluation, web-gui-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06140-curateevo-data-curation-evolving-for-agentic-post-training.md", + "title": "\"CurateEvo: Data-Curation Evolving for Agentic Post-Training\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"CurateEvo: Data-Curation Evolving for Agentic Post-Training\"", + "authors": "Dingzirui Wang, Xuanliang Zhang, Keyan Xu, Qingfu Zhu, Wanxiang Che", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06140", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "memory", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06157-llm-agents-for-deliberative-collaboration-a-study-on-joint-decision-making-under.md", + "title": "\"LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability\"", + "authors": "Chenxu Wang, Yongkun Yang, Boyuan Du, Shiwei Lin, Huaping Liu", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06157", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "18", + "collection_queries": "llm-agent, multi-agent-llm" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06195-logichunter-testing-llm-agent-frameworks-with-an-agentic-oracle.md", + "title": "\"LogicHunter: Testing LLM Agent Frameworks with an Agentic Oracle\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"LogicHunter: Testing LLM Agent Frameworks with an Agentic Oracle\"", + "authors": "Minghui Long, Yanjie Zhao, Haoyu Wang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06195", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06223-information-gain-based-rollout-policy-optimization-an-adaptive-tree-structured-r.md", + "title": "\"Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents\"", + "authors": "Yijun Zhang, Fan Xu, Jiaxin Ding, Yule Xie, Shiqing Gao, Xin Ding, Haoxiang Zhang, Luoyi Fu, et al.", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06223", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "planning", + "rag" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "15", + "collection_queries": "llm-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06273-agenttether-graph-guided-diagnosis-and-runtime-intervention-for-reliable-llm-age.md", + "title": "\"AgentTether: Graph-Guided Diagnosis and Runtime Intervention for Reliable LLM Agent Operation\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"AgentTether: Graph-Guided Diagnosis and Runtime Intervention for Reliable LLM Agent Operation\"", + "authors": "Chenyu Zhao, Shenglin Zhang, Wenwei Gu, Yongqian Sun, Dan Pei, Chetan Bansal, Saravan Rajmohan, Minghua Ma", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06273", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "computer-use", + "memory", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "17", + "collection_queries": "llm-agent, tool-use" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06341-harnessing-code-agents-for-automatic-software-verification.md", + "title": "Harnessing Code Agents for Automatic Software Verification", + "type": "paper", + "meta": { + "type": "paper", + "title": "Harnessing Code Agents for Automatic Software Verification", + "authors": "Shuangxiang Kan, Shuanglong Kan, Sebastian Ertel", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06341", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "memory", + "rag", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.FL", + "cs.AI", + "cs.SE" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "13", + "collection_queries": "coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06411-rubench-a-repository-level-agentic-coding-benchmark-with-natively-authored-russi.md", + "title": "\"RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications\"", + "authors": "Evgeny Shilov", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06411", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.SE", + "cs.AI", + "cs.CL" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agent-evaluation, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06413-an-experimental-design-approach-to-evaluating-agentic-ai-s-autonomous-model-disc.md", + "title": "\"An Experimental Design Approach to Evaluating Agentic AI's Autonomous Model Discovery\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"An Experimental Design Approach to Evaluating Agentic AI's Autonomous Model Discovery\"", + "authors": "Hao He, Xueying Liu, Chris J. Kuhlman, Xinwei Deng", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06413", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "coding-agent", + "reasoning" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "stat.ME", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "16", + "collection_queries": "agentic-ai, coding-agent" + } + }, + { + "collection": "papers", + "path": "papers/items/2026-2607-06452-from-voting-to-agent-collaboration-answer-type-aware-llm-pipelines-for-bioasq-14.md", + "title": "\"From Voting to Agent Collaboration: Answer-Type-Aware LLM Pipelines for BioASQ 14b\"", + "type": "paper", + "meta": { + "type": "paper", + "title": "\"From Voting to Agent Collaboration: Answer-Type-Aware LLM Pipelines for BioASQ 14b\"", + "authors": "Taeyun Roh, Eunha Lee, Wonjune Jang, Sohyun Chung, Junha Jung, Jaewoo Kang", + "year": "2026", + "venue": "arXiv", + "url": "https://arxiv.org/abs/2607.06452", + "code_url": [], + "source": "arxiv", + "collected_at": "2026-07-08", + "published_at": "2026-07-07", + "updated_at": "2026-07-07", + "status": "queued", + "relevance": "high", + "topics": [ + "agent-evaluation", + "multi-agent", + "rag", + "reasoning", + "tool-use" + ], + "methods": [], + "benchmarks": [], + "models": [], + "datasets": [ + "cs.CL", + "cs.AI" + ], + "related_concepts": [], + "related_jobs": [], + "related_experiments": [], + "related_projects": [], + "collection_score": "14", + "collection_queries": "multi-agent-llm" + } + }, { "collection": "papers", "path": "papers/items/2026-agent-safety-benchmark-taxonomy.md", diff --git a/data/summary.json b/data/summary.json index 891025b..6188ee7 100644 --- a/data/summary.json +++ b/data/summary.json @@ -1,8 +1,8 @@ { - "total": 21, + "total": 989, "by_collection": { "jobs": 6, - "papers": 6, + "papers": 974, "industry": 9 }, "by_status": { @@ -11,8 +11,8 @@ "unknown": 2 }, "papers": { - "skimmed": 5, - "queued": 1 + "queued": 969, + "skimmed": 5 }, "industry": { "analyzed": 8, @@ -22,44 +22,72 @@ "top_topics": [ [ "agent-evaluation", - 8 + 837 + ], + [ + "tool-use", + 728 + ], + [ + "rag", + 460 + ], + [ + "reasoning", + 378 + ], + [ + "planning", + 347 + ], + [ + "agent-safety", + 345 + ], + [ + "memory", + 296 + ], + [ + "computer-use", + 257 + ], + [ + "multi-agent", + 245 + ], + [ + "workflow-agent", + 237 + ], + [ + "coding-agent", + 205 + ], + [ + "world-model", + 83 + ], + [ + "embodied-agent", + 66 ], [ "agent", 6 ], - [ - "memory", - 5 - ], [ "evaluation", 4 ], - [ - "coding-agent", - 4 - ], [ "agent-architecture", 3 ], - [ - "agent-safety", - 3 - ], [ "enterprise-ai", 3 ], - [ - "tool-use", - 2 - ], - [ - "workflow-agent", - 1 - ], [ "llm-infra", 1 @@ -71,34 +99,6 @@ [ "harness", 1 - ], - [ - "context-compression", - 1 - ], - [ - "agent-productization", - 1 - ], - [ - "agent-infra", - 1 - ], - [ - "agent-testing", - 1 - ], - [ - "harness-engineering", - 1 - ], - [ - "human-in-the-loop", - 1 - ], - [ - "gui-agent", - 1 ] ], "top_skills": [ diff --git a/papers/README.md b/papers/README.md index a95766c..7c6ecfd 100644 --- a/papers/README.md +++ b/papers/README.md @@ -13,6 +13,7 @@ - [source-registry](source-registry.md): 来源、搜索方式和筛选规则 - [reading-queue](reading-queue.md): 阅读队列和优先级 - [paper-insights](paper-insights.md): 周期性研究洞察 +- [corpus-summary-2026-07-08](corpus-summary-2026-07-08.md): 最近一年 arXiv Agent 论文扩召回摘要 ## Collection Questions diff --git a/papers/corpus-summary-2026-07-08.md b/papers/corpus-summary-2026-07-08.md new file mode 100644 index 0000000..794210d --- /dev/null +++ b/papers/corpus-summary-2026-07-08.md @@ -0,0 +1,143 @@ +# Paper Corpus Summary: 2025-07-08 to 2026-07-08 + +status: generated-summary + +## Scope + +- source: arXiv API +- window: 2025-07-08 to 2026-07-08 +- query groups: LLM agent, language agent, AI agent, agentic AI, evaluation, memory, tool use, coding agent, web/GUI/computer-use, multi-agent LLM, safety, RAG, planning, autonomous agent +- unique candidates seen: 1506 +- high-relevance candidates selected into `papers/items/`: 970 +- total paper items after expansion: 974 + +## Data Files + +- full candidate manifest: `data/arxiv-agent-candidates-2025-07-08-to-2026-07-08.json` +- high-relevance paper manifest: `data/arxiv-agent-papers-2025-07-08-to-2026-07-08.json` +- full knowledge index: `data/index.json` +- summary index: `data/summary.json` + +## Selected Corpus Distribution + +### By Month + +| Month | Count | +| --- | ---: | +| 2025-07 | 2 | +| 2025-08 | 4 | +| 2025-09 | 8 | +| 2025-10 | 8 | +| 2025-11 | 4 | +| 2025-12 | 5 | +| 2026-01 | 6 | +| 2026-02 | 28 | +| 2026-03 | 29 | +| 2026-04 | 48 | +| 2026-05 | 181 | +| 2026-06 | 490 | +| 2026-07 | 157 | + +This distribution is partly query and recency biased, but the May-July 2026 density still strongly suggests rapid acceleration in Agent papers. The full candidate pool shows the same shape: 249 in 2026-05, 785 in 2026-06, and 265 in 2026-07. + +### By Primary Query Group + +| Query Group | Selected Count | +| --- | ---: | +| agent-memory | 103 | +| agent-safety | 96 | +| agent-evaluation | 93 | +| planning-agent | 77 | +| language-agent | 71 | +| agentic-ai | 70 | +| function-calling | 66 | +| autonomous-agent-llm | 63 | +| multi-agent-llm | 63 | +| llm-agent | 62 | +| rag-agent | 46 | +| coding-agent | 45 | +| ai-agent | 41 | +| tool-use | 37 | +| web-gui-agent | 37 | + +### By Topic Tag + +Auto tags are intentionally broad and should be treated as recall-oriented. + +| Topic | Count | +| --- | ---: | +| agent-evaluation | 831 | +| tool-use | 727 | +| rag | 462 | +| reasoning | 379 | +| planning | 348 | +| agent-safety | 343 | +| memory | 292 | +| computer-use | 256 | +| multi-agent | 245 | +| workflow-agent | 236 | +| coding-agent | 202 | +| world-model | 82 | +| embodied-agent | 66 | + +## Trend Readout + +### 1. Evaluation Is the Center of Gravity + +The strongest corpus signal is not a single Agent architecture. It is evaluation: benchmarks, trajectory auditing, failure diagnosis, safety testing, harness effects, and long-horizon task scoring. + +Implication: our knowledge base needs an `agent-evaluation` track that covers final-state success, trajectory-level correctness, failure localization, cost/latency, user interaction quality, and safety. + +### 2. Memory Became a Systems Problem + +The memory cluster is large and diverse: active memory, selective retention, memory poisoning, memory governance, GUI memory, cross-episode memory, procedural memory, and long-horizon state management. + +Implication: treat memory as an architecture subsystem with write policy, read policy, retention, forgetting, provenance, safety, and evaluation. Do not reduce it to vector search. + +### 3. Safety Shifted From Refusal to Runtime Risk + +Agent safety papers now focus on tool-use leakage, data injection, prompt injection, memory poisoning, action-boundary violations, governance, and runtime controls. + +Implication: safety docs should cover permissions, sandboxing, irreversible action confirmation, tool schemas, provenance, and runtime intervention. + +### 4. Coding Agents Are Moving Beyond SWE-Bench + +The coding-agent cluster includes long-horizon maintenance, repository-level coding, compiler optimization, GUI/terminal agents, action-boundary violations, code memory, and benchmark variants. + +Implication: coding-agent evaluation should include repository exploration, multi-file edits, release-note implementation, dialogue clarification, build/test loops, and failure attribution. + +### 5. Computer-Use and GUI Agents Are Becoming Their Own Layer + +The corpus includes OS/macOS/GUI/mobile/browser/computer-use benchmarks, visual state comparison, task-state representation, and scientific instrument control. + +Implication: computer-use agents need their own tool taxonomy, safety checklist, and benchmark map. + +### 6. Multi-Agent Is Splitting Into Collaboration, Governance, and Shared Memory + +The multi-agent cluster is not only debate or role play. It includes shared memory governance, orchestration, role specialization, majority-vote failure, routing, and organizational behavior. + +Implication: multi-agent notes should separate collaboration protocols, shared state, security boundaries, and evaluation. + +## Immediate Reading Priorities + +P0 clusters: + +- Agent evaluation / benchmark methodology +- Memory architecture and memory evaluation +- Agent safety / runtime governance +- Coding agent long-horizon evaluation +- Computer-use / GUI agent benchmarks + +P1 clusters: + +- Multi-agent shared memory and orchestration +- RAG-agent and technical literature agents +- Domain-specific agents in healthcare, finance, energy, cybersecurity +- Embodied/world-model agents + +## Caveats + +- This is a high-recall automated arXiv sweep, not a curated bibliography. +- Auto tags are broad and may over-tag papers whose abstracts use common words like planning, action, retrieval, or benchmark. +- The selected 970 are ordered by automated collection score, not by paper quality. +- The full 1506-candidate manifest should be used as the reserve pool for future expansion. diff --git a/papers/items/2025-2507-20395-mazeeval-a-benchmark-for-testing-sequential-decision-making-in-language-models.md b/papers/items/2025-2507-20395-mazeeval-a-benchmark-for-testing-sequential-decision-making-in-language-models.md new file mode 100644 index 0000000..08f2d34 --- /dev/null +++ b/papers/items/2025-2507-20395-mazeeval-a-benchmark-for-testing-sequential-decision-making-in-language-models.md @@ -0,0 +1,61 @@ +# Paper: MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models + +--- +type: paper +title: "MazeEval: A Benchmark for Testing Sequential Decision-Making in Language Models" +authors: Hafsteinn Einarsson +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2507.20395 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-07-27 +updated_at: 2025-07-27 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, embodied-agent, reasoning +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2507.20395 diff --git a/papers/items/2025-2507-20666-mimii-agent-leveraging-llms-with-function-calling-for-relative-evaluation-of-ano.md b/papers/items/2025-2507-20666-mimii-agent-leveraging-llms-with-function-calling-for-relative-evaluation-of-ano.md new file mode 100644 index 0000000..9da3cfe --- /dev/null +++ b/papers/items/2025-2507-20666-mimii-agent-leveraging-llms-with-function-calling-for-relative-evaluation-of-ano.md @@ -0,0 +1,63 @@ +# Paper: MIMII-Agent: Leveraging LLMs with Function Calling for Relative Evaluation of Anomalous Sound Detection + +--- +type: paper +title: "MIMII-Agent: Leveraging LLMs with Function Calling for Relative Evaluation of Anomalous Sound Detection" +authors: Harsh Purohit, Tomoya Nishida, Kota Dohi, Takashi Endo, Yohei Kawaguchi +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2507.20666 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-07-28 +updated_at: 2025-07-28 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.AS + - cs.AI + - cs.LG + - cs.SD +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, rag, tool-use +- arXiv categories: eess.AS, cs.AI, cs.LG, cs.SD +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2507.20666 diff --git a/papers/items/2025-2508-07575-mcptoolbench-a-large-scale-ai-agent-model-context-protocol-mcp-tool-use-benchmar.md b/papers/items/2025-2508-07575-mcptoolbench-a-large-scale-ai-agent-model-context-protocol-mcp-tool-use-benchmar.md new file mode 100644 index 0000000..85af450 --- /dev/null +++ b/papers/items/2025-2508-07575-mcptoolbench-a-large-scale-ai-agent-model-context-protocol-mcp-tool-use-benchmar.md @@ -0,0 +1,60 @@ +# Paper: MCPToolBench++: A Large Scale AI Agent Model Context Protocol MCP Tool Use Benchmark + +--- +type: paper +title: "MCPToolBench++: A Large Scale AI Agent Model Context Protocol MCP Tool Use Benchmark" +authors: Shiqing Fan, Xichen Ding, Liang Zhang, Linjian Mo +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2508.07575 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-08-11 +updated_at: 2025-08-11 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, tool-use +- arXiv categories: cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2508.07575 diff --git a/papers/items/2025-2508-11027-hell-or-high-water-evaluating-agentic-recovery-from-external-failures.md b/papers/items/2025-2508-11027-hell-or-high-water-evaluating-agentic-recovery-from-external-failures.md new file mode 100644 index 0000000..b109e19 --- /dev/null +++ b/papers/items/2025-2508-11027-hell-or-high-water-evaluating-agentic-recovery-from-external-failures.md @@ -0,0 +1,61 @@ +# Paper: Hell or High Water: Evaluating Agentic Recovery from External Failures + +--- +type: paper +title: "Hell or High Water: Evaluating Agentic Recovery from External Failures" +authors: Andrew Wang, Sophia Hager, Adi Asija, Daniel Khashabi, Nicholas Andrews +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2508.11027 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-08-14 +updated_at: 2025-08-14 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, planning, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2508.11027 diff --git a/papers/items/2025-2508-12685-toolace-mt-non-autoregressive-generation-for-agentic-multi-turn-interaction.md b/papers/items/2025-2508-12685-toolace-mt-non-autoregressive-generation-for-agentic-multi-turn-interaction.md new file mode 100644 index 0000000..05f5f69 --- /dev/null +++ b/papers/items/2025-2508-12685-toolace-mt-non-autoregressive-generation-for-agentic-multi-turn-interaction.md @@ -0,0 +1,61 @@ +# Paper: ToolACE-MT: Non-Autoregressive Generation for Agentic Multi-Turn Interaction + +--- +type: paper +title: "ToolACE-MT: Non-Autoregressive Generation for Agentic Multi-Turn Interaction" +authors: Xingshan Zeng, Weiwen Liu, Lingzhi Wang, Liangyou Li, Fei Mi, Yasheng Wang, Lifeng Shang, Xin Jiang, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2508.12685 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-08-18 +updated_at: 2026-02-13 +status: queued +relevance: high +topics: + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: tool-use, world-model +- arXiv categories: cs.CL, cs.AI, cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2508.12685 diff --git a/papers/items/2025-2508-17094-powerchain-a-verifiable-agentic-ai-system-for-automating-distribution-grid-analy.md b/papers/items/2025-2508-17094-powerchain-a-verifiable-agentic-ai-system-for-automating-distribution-grid-analy.md new file mode 100644 index 0000000..35b3611 --- /dev/null +++ b/papers/items/2025-2508-17094-powerchain-a-verifiable-agentic-ai-system-for-automating-distribution-grid-analy.md @@ -0,0 +1,63 @@ +# Paper: PowerChain: A Verifiable Agentic AI System for Automating Distribution Grid Analyses + +--- +type: paper +title: "PowerChain: A Verifiable Agentic AI System for Automating Distribution Grid Analyses" +authors: Emmanuel O. Badmus, Peng Sang, Dimitrios Stamoulis, Amritanshu Pandey +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2508.17094 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-08-23 +updated_at: 2025-10-21 +status: queued +relevance: high +topics: + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, eess.SY +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2508.17094 diff --git a/papers/items/2025-2509-02444-appcopilot-toward-general-accurate-long-horizon-and-efficient-mobile-agent.md b/papers/items/2025-2509-02444-appcopilot-toward-general-accurate-long-horizon-and-efficient-mobile-agent.md new file mode 100644 index 0000000..163301a --- /dev/null +++ b/papers/items/2025-2509-02444-appcopilot-toward-general-accurate-long-horizon-and-efficient-mobile-agent.md @@ -0,0 +1,66 @@ +# Paper: AppCopilot: Toward General, Accurate, Long-Horizon, and Efficient Mobile Agent + +--- +type: paper +title: "AppCopilot: Toward General, Accurate, Long-Horizon, and Efficient Mobile Agent" +authors: Jingru Fan, Yufan Dang, Jingyao Wu, Huatao Li, Runde Yang, Xiyuan Yang, Yuheng Wang, Chen Qian +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2509.02444 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-09-02 +updated_at: 2025-10-17 +status: queued +relevance: high +topics: + - computer-use + - memory + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.CV + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: computer-use, memory, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL, cs.CV, cs.HC +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2509.02444 diff --git a/papers/items/2025-2509-02494-gridmind-llms-powered-agents-for-power-system-analysis-and-operations.md b/papers/items/2025-2509-02494-gridmind-llms-powered-agents-for-power-system-analysis-and-operations.md new file mode 100644 index 0000000..20b2d3a --- /dev/null +++ b/papers/items/2025-2509-02494-gridmind-llms-powered-agents-for-power-system-analysis-and-operations.md @@ -0,0 +1,60 @@ +# Paper: GridMind: LLMs-Powered Agents for Power System Analysis and Operations + +--- +type: paper +title: "GridMind: LLMs-Powered Agents for Power System Analysis and Operations" +authors: Hongwei Jin, Kibaek Kim, Jonghwan Kwon +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2509.02494 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-09-02 +updated_at: 2025-09-02 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, multi-agent, workflow-agent +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2509.02494 diff --git a/papers/items/2025-2509-08863-geojson-agents-a-multi-agent-llm-architecture-for-geospatial-analysis-function-c.md b/papers/items/2025-2509-08863-geojson-agents-a-multi-agent-llm-architecture-for-geospatial-analysis-function-c.md new file mode 100644 index 0000000..bc9e667 --- /dev/null +++ b/papers/items/2025-2509-08863-geojson-agents-a-multi-agent-llm-architecture-for-geospatial-analysis-function-c.md @@ -0,0 +1,62 @@ +# Paper: GeoJSON Agents:A Multi-Agent LLM Architecture for Geospatial Analysis-Function Calling vs Code Generation + +--- +type: paper +title: "GeoJSON Agents:A Multi-Agent LLM Architecture for Geospatial Analysis-Function Calling vs Code Generation" +authors: Qianqian Luo, Qingming Lin, Liuchang Xu, Sensen Wu, Ruichen Mao, Chao Wang, Hailin Feng, Bo Huang, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2509.08863 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-09-10 +updated_at: 2025-12-03 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, multi-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2509.08863 diff --git a/papers/items/2025-2509-10769-agentarch-a-comprehensive-benchmark-to-evaluate-agent-architectures-in-enterpris.md b/papers/items/2025-2509-10769-agentarch-a-comprehensive-benchmark-to-evaluate-agent-architectures-in-enterpris.md new file mode 100644 index 0000000..7f20ca4 --- /dev/null +++ b/papers/items/2025-2509-10769-agentarch-a-comprehensive-benchmark-to-evaluate-agent-architectures-in-enterpris.md @@ -0,0 +1,64 @@ +# Paper: AgentArch: A Comprehensive Benchmark to Evaluate Agent Architectures in Enterprise + +--- +type: paper +title: "AgentArch: A Comprehensive Benchmark to Evaluate Agent Architectures in Enterprise" +authors: Tara Bogavelli, Roshnee Sharma, Hari Subramani +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2509.10769 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-09-13 +updated_at: 2026-01-06 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, memory, multi-agent, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CL, cs.MA +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2509.10769 diff --git a/papers/items/2025-2509-13311-towards-general-agentic-intelligence-via-environment-scaling.md b/papers/items/2025-2509-13311-towards-general-agentic-intelligence-via-environment-scaling.md new file mode 100644 index 0000000..35f7969 --- /dev/null +++ b/papers/items/2025-2509-13311-towards-general-agentic-intelligence-via-environment-scaling.md @@ -0,0 +1,59 @@ +# Paper: Towards General Agentic Intelligence via Environment Scaling + +--- +type: paper +title: Towards General Agentic Intelligence via Environment Scaling +authors: Runnan Fang, Shihao Cai, Baixuan Li, Jialong Wu, Guangyu Li, Wenbiao Yin, Xinyu Wang, Xiaobin Wang, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2509.13311 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-09-16 +updated_at: 2025-09-16 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2509.13311 diff --git a/papers/items/2025-2509-14477-ticket-bench-a-kickoff-for-multilingual-and-regionalized-agent-evaluation.md b/papers/items/2025-2509-14477-ticket-bench-a-kickoff-for-multilingual-and-regionalized-agent-evaluation.md new file mode 100644 index 0000000..0e73d8b --- /dev/null +++ b/papers/items/2025-2509-14477-ticket-bench-a-kickoff-for-multilingual-and-regionalized-agent-evaluation.md @@ -0,0 +1,60 @@ +# Paper: Ticket-Bench: A Kickoff for Multilingual and Regionalized Agent Evaluation + +--- +type: paper +title: "Ticket-Bench: A Kickoff for Multilingual and Regionalized Agent Evaluation" +authors: Thales Sales Almeida, João Guilherme Alves Santos, Thiago Laitz, Giovana Kerche Bonás +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2509.14477 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-09-17 +updated_at: 2025-09-17 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, reasoning +- arXiv categories: cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2509.14477 diff --git a/papers/items/2025-2509-20998-core-full-path-evaluation-of-llm-agents-beyond-final-state.md b/papers/items/2025-2509-20998-core-full-path-evaluation-of-llm-agents-beyond-final-state.md new file mode 100644 index 0000000..822d8df --- /dev/null +++ b/papers/items/2025-2509-20998-core-full-path-evaluation-of-llm-agents-beyond-final-state.md @@ -0,0 +1,61 @@ +# Paper: CORE: Full-Path Evaluation of LLM Agents Beyond Final State + +--- +type: paper +title: "CORE: Full-Path Evaluation of LLM Agents Beyond Final State" +authors: Panagiotis Michelakis, Yiannis Hadjiyiannis, Dimitrios Stamoulis +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2509.20998 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-09-25 +updated_at: 2025-09-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2509.20998 diff --git a/papers/items/2025-2509-26553-towards-reliable-benchmarking-a-contamination-free-controllable-evaluation-frame.md b/papers/items/2025-2509-26553-towards-reliable-benchmarking-a-contamination-free-controllable-evaluation-frame.md new file mode 100644 index 0000000..e0d18f8 --- /dev/null +++ b/papers/items/2025-2509-26553-towards-reliable-benchmarking-a-contamination-free-controllable-evaluation-frame.md @@ -0,0 +1,61 @@ +# Paper: Towards Reliable Benchmarking: A Contamination Free, Controllable Evaluation Framework for Multi-step LLM Function Calling + +--- +type: paper +title: "Towards Reliable Benchmarking: A Contamination Free, Controllable Evaluation Framework for Multi-step LLM Function Calling" +authors: Seiji Maekawa, Jackson Hassell, Pouya Pezeshkpour, Tom Mitchell, Estevam Hruschka +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2509.26553 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-09-30 +updated_at: 2026-02-06 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.PL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, reasoning, tool-use +- arXiv categories: cs.CL, cs.PL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2509.26553 diff --git a/papers/items/2025-2510-03847-small-language-models-for-agentic-systems-a-survey-of-architectures-capabilities.md b/papers/items/2025-2510-03847-small-language-models-for-agentic-systems-a-survey-of-architectures-capabilities.md new file mode 100644 index 0000000..aa4b88a --- /dev/null +++ b/papers/items/2025-2510-03847-small-language-models-for-agentic-systems-a-survey-of-architectures-capabilities.md @@ -0,0 +1,65 @@ +# Paper: Small Language Models for Agentic Systems: A Survey of Architectures, Capabilities, and Deployment Trade offs + +--- +type: paper +title: "Small Language Models for Agentic Systems: A Survey of Architectures, Capabilities, and Deployment Trade offs" +authors: Raghav Sharma, Manan Mehta +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2510.03847 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-10-04 +updated_at: 2025-10-04 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, coding-agent, computer-use, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.LG +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2510.03847 diff --git a/papers/items/2025-2510-04206-agentrl-scaling-agentic-reinforcement-learning-with-a-multi-turn-multi-task-fram.md b/papers/items/2025-2510-04206-agentrl-scaling-agentic-reinforcement-learning-with-a-multi-turn-multi-task-fram.md new file mode 100644 index 0000000..0ede631 --- /dev/null +++ b/papers/items/2025-2510-04206-agentrl-scaling-agentic-reinforcement-learning-with-a-multi-turn-multi-task-fram.md @@ -0,0 +1,59 @@ +# Paper: AgentRL: Scaling Agentic Reinforcement Learning with a Multi-Turn, Multi-Task Framework + +--- +type: paper +title: "AgentRL: Scaling Agentic Reinforcement Learning with a Multi-Turn, Multi-Task Framework" +authors: Hanchen Zhang, Xiao Liu, Bowen Lv, Xueqiao Sun, Bohao Jing, Iat Long Iong, Zhenyu Hou, Zehan Qi, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2510.04206 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-10-05 +updated_at: 2025-10-05 +status: queued +relevance: high +topics: + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: rag, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2510.04206 diff --git a/papers/items/2025-2510-14548-llm-agents-beyond-utility-an-open-ended-perspective.md b/papers/items/2025-2510-14548-llm-agents-beyond-utility-an-open-ended-perspective.md new file mode 100644 index 0000000..ae8851e --- /dev/null +++ b/papers/items/2025-2510-14548-llm-agents-beyond-utility-an-open-ended-perspective.md @@ -0,0 +1,61 @@ +# Paper: LLM Agents Beyond Utility: An Open-Ended Perspective + +--- +type: paper +title: "LLM Agents Beyond Utility: An Open-Ended Perspective" +authors: Asen Nachkov, Xi Wang, Luc Van Gool +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2510.14548 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-10-16 +updated_at: 2025-10-16 +status: queued +relevance: high +topics: + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: memory, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2510.14548 diff --git a/papers/items/2025-2510-18586-tokencake-a-kv-cache-centric-serving-framework-for-llm-based-multi-agent-applica.md b/papers/items/2025-2510-18586-tokencake-a-kv-cache-centric-serving-framework-for-llm-based-multi-agent-applica.md new file mode 100644 index 0000000..c525ce5 --- /dev/null +++ b/papers/items/2025-2510-18586-tokencake-a-kv-cache-centric-serving-framework-for-llm-based-multi-agent-applica.md @@ -0,0 +1,61 @@ +# Paper: TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications + +--- +type: paper +title: "TokenCake: A KV-Cache-centric Serving Framework for LLM-based Multi-Agent Applications" +authors: Zhuohang Bian, Feiyang Wu, Zhuoran Li, Teng Ma, Youwei Zhuo +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2510.18586 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-10-21 +updated_at: 2026-05-20 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, memory, multi-agent +- arXiv categories: cs.DC +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2510.18586 diff --git a/papers/items/2025-2510-21524-eu-agent-bench-measuring-illegal-behavior-of-llm-agents-under-eu-law.md b/papers/items/2025-2510-21524-eu-agent-bench-measuring-illegal-behavior-of-llm-agents-under-eu-law.md new file mode 100644 index 0000000..fc5ea5b --- /dev/null +++ b/papers/items/2025-2510-21524-eu-agent-bench-measuring-illegal-behavior-of-llm-agents-under-eu-law.md @@ -0,0 +1,61 @@ +# Paper: EU-Agent-Bench: Measuring Illegal Behavior of LLM Agents Under EU Law + +--- +type: paper +title: "EU-Agent-Bench: Measuring Illegal Behavior of LLM Agents Under EU Law" +authors: Ilija Lichkovski, Alexander Müller, Mariam Ibrahim, Tiwai Mhundwa +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2510.21524 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-10-24 +updated_at: 2025-10-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2510.21524 diff --git a/papers/items/2025-2510-22768-seeing-is-believing-evaluating-vision-language-model-susceptibility-in-agent-to-.md b/papers/items/2025-2510-22768-seeing-is-believing-evaluating-vision-language-model-susceptibility-in-agent-to-.md new file mode 100644 index 0000000..0e52ea5 --- /dev/null +++ b/papers/items/2025-2510-22768-seeing-is-believing-evaluating-vision-language-model-susceptibility-in-agent-to-.md @@ -0,0 +1,62 @@ +# Paper: Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion + +--- +type: paper +title: Seeing is Believing? Evaluating Vision-Language Model Susceptibility in Agent-to-Agent Multimodal Persuasion +authors: Haoyi Qiu, Yilun Zhou, Pranav Narayanan Venkit, Kung-Hsiang Huang, Jiaxin Zhang, Nanyun Peng, Chien-Sheng Wu +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2510.22768 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-10-26 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, tool-use +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2510.22768 diff --git a/papers/items/2025-2510-24645-funreason-mt-technical-report-advanced-data-synthesis-solution-for-real-world-mu.md b/papers/items/2025-2510-24645-funreason-mt-technical-report-advanced-data-synthesis-solution-for-real-world-mu.md new file mode 100644 index 0000000..1ae8e05 --- /dev/null +++ b/papers/items/2025-2510-24645-funreason-mt-technical-report-advanced-data-synthesis-solution-for-real-world-mu.md @@ -0,0 +1,61 @@ +# Paper: FunReason-MT Technical Report: Advanced Data Synthesis Solution for Real-world Multi-Turn Tool-use + +--- +type: paper +title: "FunReason-MT Technical Report: Advanced Data Synthesis Solution for Real-world Multi-Turn Tool-use" +authors: Zengzhuang Xu, Bingguang Hao, Zechuan Wang, Yuntao Wen, Xinyi Xu, Yang Liu, Long Chen, Dong Wang, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2510.24645 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-10-28 +updated_at: 2025-11-16 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, multi-agent, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2510.24645 diff --git a/papers/items/2025-2510-26167-toolrm-towards-agentic-tool-use-reward-modeling.md b/papers/items/2025-2510-26167-toolrm-towards-agentic-tool-use-reward-modeling.md new file mode 100644 index 0000000..eaa8e1d --- /dev/null +++ b/papers/items/2025-2510-26167-toolrm-towards-agentic-tool-use-reward-modeling.md @@ -0,0 +1,60 @@ +# Paper: ToolRM: Towards Agentic Tool-Use Reward Modeling + +--- +type: paper +title: "ToolRM: Towards Agentic Tool-Use Reward Modeling" +authors: Renhao Li, Jianhong Tu, Yang Su, Yantao Liu, Fei Huang, Hamid Alinejad-Rokny, Derek F. Wong, Junyang Lin, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2510.26167 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-10-30 +updated_at: 2026-01-13 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2510.26167 diff --git a/papers/items/2025-2511-04847-test-time-adaptation-for-llm-agents-via-environment-interaction.md b/papers/items/2025-2511-04847-test-time-adaptation-for-llm-agents-via-environment-interaction.md new file mode 100644 index 0000000..32c1304 --- /dev/null +++ b/papers/items/2025-2511-04847-test-time-adaptation-for-llm-agents-via-environment-interaction.md @@ -0,0 +1,63 @@ +# Paper: Test-Time Adaptation for LLM Agents via Environment Interaction + +--- +type: paper +title: Test-Time Adaptation for LLM Agents via Environment Interaction +authors: Arthur Chen, Zuxin Liu, Jianguo Zhang, Akshara Prabhakar, Zhiwei Liu, Shelby Heinecke, Silvio Savarese, Victor Zhong, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2511.04847 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-11-06 +updated_at: 2026-02-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - embodied-agent + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, embodied-agent, rag, tool-use, world-model +- arXiv categories: cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2511.04847 diff --git a/papers/items/2025-2511-11169-refine-and-align-confidence-calibration-through-multi-agent-interaction-in-vqa.md b/papers/items/2025-2511-11169-refine-and-align-confidence-calibration-through-multi-agent-interaction-in-vqa.md new file mode 100644 index 0000000..36d5f53 --- /dev/null +++ b/papers/items/2025-2511-11169-refine-and-align-confidence-calibration-through-multi-agent-interaction-in-vqa.md @@ -0,0 +1,63 @@ +# Paper: Refine and Align: Confidence Calibration through Multi-Agent Interaction in VQA + +--- +type: paper +title: "Refine and Align: Confidence Calibration through Multi-Agent Interaction in VQA" +authors: Ayush Pandey, Jai Bardhan, Ishita Jain, Ramya S Hebbalaguppe, Rohan Raju Dhanakshirur, Lovekesh Vig +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2511.11169 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-11-14 +updated_at: 2025-11-14 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, embodied-agent, multi-agent, tool-use +- arXiv categories: cs.CV, cs.AI, cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2511.11169 diff --git a/papers/items/2025-2511-15203-taxonomy-evaluation-and-exploitation-of-ipi-centric-llm-agent-defense-frameworks.md b/papers/items/2025-2511-15203-taxonomy-evaluation-and-exploitation-of-ipi-centric-llm-agent-defense-frameworks.md new file mode 100644 index 0000000..16acdbf --- /dev/null +++ b/papers/items/2025-2511-15203-taxonomy-evaluation-and-exploitation-of-ipi-centric-llm-agent-defense-frameworks.md @@ -0,0 +1,62 @@ +# Paper: Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks + +--- +type: paper +title: Taxonomy, Evaluation and Exploitation of IPI-Centric LLM Agent Defense Frameworks +authors: Zimo Ji, Xunguang Wang, Zongjie Li, Pingchuan Ma, Yudong Gao, Daoyuan Wu, Xincheng Yan, Tian Tian, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2511.15203 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-11-19 +updated_at: 2025-11-19 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2511.15203 diff --git a/papers/items/2025-2511-22138-tinyllm-evaluation-and-optimization-of-small-language-models-for-agentic-tasks-o.md b/papers/items/2025-2511-22138-tinyllm-evaluation-and-optimization-of-small-language-models-for-agentic-tasks-o.md new file mode 100644 index 0000000..5661519 --- /dev/null +++ b/papers/items/2025-2511-22138-tinyllm-evaluation-and-optimization-of-small-language-models-for-agentic-tasks-o.md @@ -0,0 +1,60 @@ +# Paper: TinyLLM: Evaluation and Optimization of Small Language Models for Agentic Tasks on Edge Devices + +--- +type: paper +title: "TinyLLM: Evaluation and Optimization of Small Language Models for Agentic Tasks on Edge Devices" +authors: Mohd Ariful Haque, Fahad Rahman, Kishor Datta Gupta, Khalil Shujaee, Roy George +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2511.22138 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-11-27 +updated_at: 2025-11-27 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2511.22138 diff --git a/papers/items/2025-2512-02605-iact-a-self-organizing-recursive-model-for-general-ai-agents-a-technical-white-p.md b/papers/items/2025-2512-02605-iact-a-self-organizing-recursive-model-for-general-ai-agents-a-technical-white-p.md new file mode 100644 index 0000000..f57e13a --- /dev/null +++ b/papers/items/2025-2512-02605-iact-a-self-organizing-recursive-model-for-general-ai-agents-a-technical-white-p.md @@ -0,0 +1,65 @@ +# Paper: IACT: A Self-Organizing Recursive Model for General AI Agents: A Technical White Paper on the Architecture Behind kragent.ai + +--- +type: paper +title: "IACT: A Self-Organizing Recursive Model for General AI Agents: A Technical White Paper on the Architecture Behind kragent.ai" +authors: Pengju Lu +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2512.02605 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-12-02 +updated_at: 2025-12-02 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, memory, rag, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.MA, cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2512.02605 diff --git a/papers/items/2025-2512-11682-medai-evaluating-txagent-s-therapeutic-agentic-reasoning-in-the-neurips-cure-ben.md b/papers/items/2025-2512-11682-medai-evaluating-txagent-s-therapeutic-agentic-reasoning-in-the-neurips-cure-ben.md new file mode 100644 index 0000000..6a031df --- /dev/null +++ b/papers/items/2025-2512-11682-medai-evaluating-txagent-s-therapeutic-agentic-reasoning-in-the-neurips-cure-ben.md @@ -0,0 +1,65 @@ +# Paper: MedAI: Evaluating TxAgent's Therapeutic Agentic Reasoning in the NeurIPS CURE-Bench Competition + +--- +type: paper +title: "MedAI: Evaluating TxAgent's Therapeutic Agentic Reasoning in the NeurIPS CURE-Bench Competition" +authors: Tim Cofala, Christian Kalfar, Jingge Xiao, Johanna Schrader, Michelle Tang, Wolfgang Nejdl +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2512.11682 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-12-12 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, computer-use, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2512.11682 diff --git a/papers/items/2025-2512-23611-close-the-loop-synthesizing-infinite-tool-use-data-via-multi-agent-role-playing.md b/papers/items/2025-2512-23611-close-the-loop-synthesizing-infinite-tool-use-data-via-multi-agent-role-playing.md new file mode 100644 index 0000000..a93c320 --- /dev/null +++ b/papers/items/2025-2512-23611-close-the-loop-synthesizing-infinite-tool-use-data-via-multi-agent-role-playing.md @@ -0,0 +1,62 @@ +# Paper: Close the Loop: Synthesizing Infinite Tool-Use Data via Multi-Agent Role-Playing + +--- +type: paper +title: "Close the Loop: Synthesizing Infinite Tool-Use Data via Multi-Agent Role-Playing" +authors: Yuwen Li, Wei Zhang, Zelong Huang, Mason Yang, Jiajun Wu, Shawn Guo, Huahao Hu, Lingyi Sun, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2512.23611 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-12-29 +updated_at: 2025-12-29 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, multi-agent, rag, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2512.23611 diff --git a/papers/items/2025-2512-23647-nested-browser-use-learning-for-agentic-information-seeking.md b/papers/items/2025-2512-23647-nested-browser-use-learning-for-agentic-information-seeking.md new file mode 100644 index 0000000..b77ac95 --- /dev/null +++ b/papers/items/2025-2512-23647-nested-browser-use-learning-for-agentic-information-seeking.md @@ -0,0 +1,65 @@ +# Paper: Nested Browser-Use Learning for Agentic Information Seeking + +--- +type: paper +title: Nested Browser-Use Learning for Agentic Information Seeking +authors: Baixuan Li, Jialong Wu, Wenbiao Yin, Kuan Li, Zhongwang Zhang, Huifeng Yin, Zhengwei Tao, Liwen Zhang, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2512.23647 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-12-29 +updated_at: 2025-12-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.IR + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, rag, reasoning, tool-use +- arXiv categories: cs.CL, cs.AI, cs.IR, cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2512.23647 diff --git a/papers/items/2025-2512-23747-state-of-the-art-small-language-coder-model-mify-coder.md b/papers/items/2025-2512-23747-state-of-the-art-small-language-coder-model-mify-coder.md new file mode 100644 index 0000000..9d6a861 --- /dev/null +++ b/papers/items/2025-2512-23747-state-of-the-art-small-language-coder-model-mify-coder.md @@ -0,0 +1,64 @@ +# Paper: State-of-the-art Small Language Coder Model: Mify-Coder + +--- +type: paper +title: "State-of-the-art Small Language Coder Model: Mify-Coder" +authors: Abhinav Parmar, Abhisek Panigrahi, Abhishek Kumar Dwivedi, Abhishek Bhattacharya, Adarsh Ramachandra, Aditya Choudhary, Aditya Garg, Aditya Raj, et al. +year: 2025 +venue: arXiv +url: https://arxiv.org/abs/2512.23747 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2025-12-26 +updated_at: 2025-12-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, workflow-agent +- arXiv categories: cs.SE, cs.AI, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2512.23747 diff --git a/papers/items/2026-2601-00268-beyond-perfect-apis-a-comprehensive-evaluation-of-llm-agents-under-real-world-ap.md b/papers/items/2026-2601-00268-beyond-perfect-apis-a-comprehensive-evaluation-of-llm-agents-under-real-world-ap.md new file mode 100644 index 0000000..e73c5b8 --- /dev/null +++ b/papers/items/2026-2601-00268-beyond-perfect-apis-a-comprehensive-evaluation-of-llm-agents-under-real-world-ap.md @@ -0,0 +1,60 @@ +# Paper: Beyond Perfect APIs: A Comprehensive Evaluation of LLM Agents Under Real-World API Complexity + +--- +type: paper +title: "Beyond Perfect APIs: A Comprehensive Evaluation of LLM Agents Under Real-World API Complexity" +authors: Doyoung Kim, Zhiwei Ren, Jie Hao, Zhongkai Sun, Lichao Wang, Xiyao Ma, Zack Ye, Xu Han, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2601.00268 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-01-01 +updated_at: 2026-01-01 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2601.00268 diff --git a/papers/items/2026-2601-05467-stelp-secure-transpilation-and-execution-of-llm-generated-programs.md b/papers/items/2026-2601-05467-stelp-secure-transpilation-and-execution-of-llm-generated-programs.md new file mode 100644 index 0000000..9856897 --- /dev/null +++ b/papers/items/2026-2601-05467-stelp-secure-transpilation-and-execution-of-llm-generated-programs.md @@ -0,0 +1,64 @@ +# Paper: STELP: Secure Transpilation and Execution of LLM-Generated Programs + +--- +type: paper +title: "STELP: Secure Transpilation and Execution of LLM-Generated Programs" +authors: Swapnil Shinde, Sahil Wadhwa, Andy Luo, Akshay Gupta, Mohammad Shahed Sorower +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2601.05467 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-01-09 +updated_at: 2026-01-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2601.05467 diff --git a/papers/items/2026-2601-06007-don-t-break-the-cache-an-evaluation-of-prompt-caching-for-long-horizon-agentic-t.md b/papers/items/2026-2601-06007-don-t-break-the-cache-an-evaluation-of-prompt-caching-for-long-horizon-agentic-t.md new file mode 100644 index 0000000..9117dfc --- /dev/null +++ b/papers/items/2026-2601-06007-don-t-break-the-cache-an-evaluation-of-prompt-caching-for-long-horizon-agentic-t.md @@ -0,0 +1,61 @@ +# Paper: Don't Break the Cache: An Evaluation of Prompt Caching for Long-Horizon Agentic Tasks + +--- +type: paper +title: "Don't Break the Cache: An Evaluation of Prompt Caching for Long-Horizon Agentic Tasks" +authors: Elias Lumer, Faheem Nizar, Akshaya Jangiti, Kevin Frank, Anmol Gulati, Mandar Phadate, Vamse Kumar Subbiah +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2601.06007 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-01-09 +updated_at: 2026-01-31 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, planning, tool-use +- arXiv categories: cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2601.06007 diff --git a/papers/items/2026-2601-06606-cedar-context-engineering-for-agentic-data-science.md b/papers/items/2026-2601-06606-cedar-context-engineering-for-agentic-data-science.md new file mode 100644 index 0000000..0f46997 --- /dev/null +++ b/papers/items/2026-2601-06606-cedar-context-engineering-for-agentic-data-science.md @@ -0,0 +1,60 @@ +# Paper: CEDAR: Context Engineering for Agentic Data Science + +--- +type: paper +title: "CEDAR: Context Engineering for Agentic Data Science" +authors: Rishiraj Saha Roy, Chris Hinze, Luzian Hahn, Fabian Kuech +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2601.06606 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-01-10 +updated_at: 2026-04-22 +status: queued +relevance: high +topics: + - planning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: planning, workflow-agent +- arXiv categories: cs.LG, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2601.06606 diff --git a/papers/items/2026-2601-12988-paperguide-making-small-language-model-paper-reading-agents-more-efficient.md b/papers/items/2026-2601-12988-paperguide-making-small-language-model-paper-reading-agents-more-efficient.md new file mode 100644 index 0000000..5b0276e --- /dev/null +++ b/papers/items/2026-2601-12988-paperguide-making-small-language-model-paper-reading-agents-more-efficient.md @@ -0,0 +1,62 @@ +# Paper: PaperGuide: Making Small Language-Model Paper-Reading Agents More Efficient + +--- +type: paper +title: "PaperGuide: Making Small Language-Model Paper-Reading Agents More Efficient" +authors: Zijian Wang, Tiancheng Huang, Hanqi Li, Da Ma, Lu Chen, Kai Yu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2601.12988 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-01-19 +updated_at: 2026-01-19 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, planning, reasoning, tool-use +- arXiv categories: cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2601.12988 diff --git a/papers/items/2026-2601-14652-mas-orchestra-understanding-and-improving-multi-agent-reasoning-through-holistic.md b/papers/items/2026-2601-14652-mas-orchestra-understanding-and-improving-multi-agent-reasoning-through-holistic.md new file mode 100644 index 0000000..e2db118 --- /dev/null +++ b/papers/items/2026-2601-14652-mas-orchestra-understanding-and-improving-multi-agent-reasoning-through-holistic.md @@ -0,0 +1,63 @@ +# Paper: MAS-Orchestra: Understanding and Improving Multi-Agent Reasoning Through Holistic Orchestration and Controlled Benchmarks + +--- +type: paper +title: "MAS-Orchestra: Understanding and Improving Multi-Agent Reasoning Through Holistic Orchestration and Controlled Benchmarks" +authors: Zixuan Ke, Yifei Ming, Austin Xu, Ryan Chin, Xuan-Phi Nguyen, Prathyusha Jwalapuram, Jiayu Wang, Semih Yavuz, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2601.14652 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-01-21 +updated_at: 2026-05-21 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, multi-agent, reasoning +- arXiv categories: cs.AI, cs.CL, cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2601.14652 diff --git a/papers/items/2026-2602-03117-agentdyn-are-your-agent-security-defenses-deployable-in-real-world-dynamic-envir.md b/papers/items/2026-2602-03117-agentdyn-are-your-agent-security-defenses-deployable-in-real-world-dynamic-envir.md new file mode 100644 index 0000000..282879b --- /dev/null +++ b/papers/items/2026-2602-03117-agentdyn-are-your-agent-security-defenses-deployable-in-real-world-dynamic-envir.md @@ -0,0 +1,61 @@ +# Paper: AgentDyn: Are Your Agent Security Defenses Deployable in Real-World Dynamic Environments? + +--- +type: paper +title: "AgentDyn: Are Your Agent Security Defenses Deployable in Real-World Dynamic Environments?" +authors: Hao Li, Ruoyao Wen, Shanghao Shi, Ning Zhang, Yevgeniy Vorobeychik, Chaowei Xiao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.03117 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-03 +updated_at: 2026-05-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, planning, tool-use +- arXiv categories: cs.CR +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.03117 diff --git a/papers/items/2026-2602-03224-tame-a-trustworthy-test-time-evolution-of-agent-memory-with-systematic-benchmark.md b/papers/items/2026-2602-03224-tame-a-trustworthy-test-time-evolution-of-agent-memory-with-systematic-benchmark.md new file mode 100644 index 0000000..4d309cb --- /dev/null +++ b/papers/items/2026-2602-03224-tame-a-trustworthy-test-time-evolution-of-agent-memory-with-systematic-benchmark.md @@ -0,0 +1,63 @@ +# Paper: TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking + +--- +type: paper +title: "TAME: A Trustworthy Test-Time Evolution of Agent Memory with Systematic Benchmarking" +authors: Yu Cheng, Yongkang Hu, Jiuan Zhou, Yushuo Zhang, Yihang Chen, Huichi Zhou, Mingang Chen, Zhizhong Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.03224 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-03 +updated_at: 2026-06-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, reasoning +- arXiv categories: cs.AI, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.03224 diff --git a/papers/items/2026-2602-03786-aorchestra-automating-sub-agent-creation-for-agentic-orchestration.md b/papers/items/2026-2602-03786-aorchestra-automating-sub-agent-creation-for-agentic-orchestration.md new file mode 100644 index 0000000..e802a08 --- /dev/null +++ b/papers/items/2026-2602-03786-aorchestra-automating-sub-agent-creation-for-agentic-orchestration.md @@ -0,0 +1,63 @@ +# Paper: AOrchestra: Automating Sub-Agent Creation for Agentic Orchestration + +--- +type: paper +title: "AOrchestra: Automating Sub-Agent Creation for Agentic Orchestration" +authors: Jianhao Ruan, Zhihao Xu, Yiran Peng, Fashen Ren, Zhaoyang Yu, Xinbing Liang, Jinyu Xiang, Yongru Chen, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.03786 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-03 +updated_at: 2026-02-07 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, coding-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.03786 diff --git a/papers/items/2026-2602-05115-socialveil-probing-social-intelligence-of-language-agents-under-communication-ba.md b/papers/items/2026-2602-05115-socialveil-probing-social-intelligence-of-language-agents-under-communication-ba.md new file mode 100644 index 0000000..88c66e6 --- /dev/null +++ b/papers/items/2026-2602-05115-socialveil-probing-social-intelligence-of-language-agents-under-communication-ba.md @@ -0,0 +1,61 @@ +# Paper: SocialVeil: Probing Social Intelligence of Language Agents under Communication Barriers + +--- +type: paper +title: "SocialVeil: Probing Social Intelligence of Language Agents under Communication Barriers" +authors: Keyang Xuan, Pengda Wang, Chongrui Ye, Haofei Yu, Tal August, Jiaxuan You +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.05115 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-04 +updated_at: 2026-02-04 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, rag, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.05115 diff --git a/papers/items/2026-2602-05302-piearena-ranking-and-profiling-language-agents-in-realistic-negotiation-scenario.md b/papers/items/2026-2602-05302-piearena-ranking-and-profiling-language-agents-in-realistic-negotiation-scenario.md new file mode 100644 index 0000000..b4f8a30 --- /dev/null +++ b/papers/items/2026-2602-05302-piearena-ranking-and-profiling-language-agents-in-realistic-negotiation-scenario.md @@ -0,0 +1,61 @@ +# Paper: PieArena: Ranking and Profiling Language Agents in Realistic Negotiation Scenarios + +--- +type: paper +title: "PieArena: Ranking and Profiling Language Agents in Realistic Negotiation Scenarios" +authors: Chris Zhu, Sasha Cui, Will Sanok Dufallo, Runzhi Jin, Zhen Xu, Linjun Zhang, Daylian Cain +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.05302 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-05 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.05302 diff --git a/papers/items/2026-2602-05386-spider-sense-intrinsic-risk-sensing-for-efficient-agent-defense-with-hierarchica.md b/papers/items/2026-2602-05386-spider-sense-intrinsic-risk-sensing-for-efficient-agent-defense-with-hierarchica.md new file mode 100644 index 0000000..247d60b --- /dev/null +++ b/papers/items/2026-2602-05386-spider-sense-intrinsic-risk-sensing-for-efficient-agent-defense-with-hierarchica.md @@ -0,0 +1,62 @@ +# Paper: Spider-Sense: Intrinsic Risk Sensing for Efficient Agent Defense with Hierarchical Adaptive Screening + +--- +type: paper +title: "Spider-Sense: Intrinsic Risk Sensing for Efficient Agent Defense with Hierarchical Adaptive Screening" +authors: Zhenxiong Yu, Zhi Yang, Zhiheng Jin, Shuhe Wang, Heng Zhang, Yanlin Fei, Lingfeng Zeng, Fangqi Lou, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.05386 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-05 +updated_at: 2026-02-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, reasoning, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.05386 diff --git a/papers/items/2026-2602-07391-naamse-framework-for-evolutionary-security-evaluation-of-agents.md b/papers/items/2026-2602-07391-naamse-framework-for-evolutionary-security-evaluation-of-agents.md new file mode 100644 index 0000000..2624b6b --- /dev/null +++ b/papers/items/2026-2602-07391-naamse-framework-for-evolutionary-security-evaluation-of-agents.md @@ -0,0 +1,60 @@ +# Paper: NAAMSE: Framework for Evolutionary Security Evaluation of Agents + +--- +type: paper +title: "NAAMSE: Framework for Evolutionary Security Evaluation of Agents" +authors: Kunal Pai, Parth Shah, Harshil Patel +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.07391 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-07 +updated_at: 2026-03-08 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety +- arXiv categories: cs.AI, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.07391 diff --git a/papers/items/2026-2602-07652-agent-fence-mapping-security-vulnerabilities-across-deep-research-agents.md b/papers/items/2026-2602-07652-agent-fence-mapping-security-vulnerabilities-across-deep-research-agents.md new file mode 100644 index 0000000..2223e54 --- /dev/null +++ b/papers/items/2026-2602-07652-agent-fence-mapping-security-vulnerabilities-across-deep-research-agents.md @@ -0,0 +1,64 @@ +# Paper: Agent-Fence: Mapping Security Vulnerabilities Across Deep Research Agents + +--- +type: paper +title: "Agent-Fence: Mapping Security Vulnerabilities Across Deep Research Agents" +authors: Sai Puppala, Ismail Hossain, Md Jahangir Alam, Yoonpyo Lee, Jay Yoo, Tanzim Ahad, Syed Bahauddin Alam, Sajedul Talukder +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.07652 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-07 +updated_at: 2026-02-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, memory, planning, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.07652 diff --git a/papers/items/2026-2602-07962-loca-bench-benchmarking-language-agents-under-controllable-and-extreme-context-g.md b/papers/items/2026-2602-07962-loca-bench-benchmarking-language-agents-under-controllable-and-extreme-context-g.md new file mode 100644 index 0000000..79ec15b --- /dev/null +++ b/papers/items/2026-2602-07962-loca-bench-benchmarking-language-agents-under-controllable-and-extreme-context-g.md @@ -0,0 +1,61 @@ +# Paper: LOCA-bench: Benchmarking Language Agents Under Controllable and Extreme Context Growth + +--- +type: paper +title: "LOCA-bench: Benchmarking Language Agents Under Controllable and Extreme Context Growth" +authors: Weihao Zeng, Yuzhen Huang, Junxian He +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.07962 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-08 +updated_at: 2026-02-08 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.07962 diff --git a/papers/items/2026-2602-08082-spectral-guardrails-for-agents-in-the-wild-detecting-tool-use-hallucinations-via.md b/papers/items/2026-2602-08082-spectral-guardrails-for-agents-in-the-wild-detecting-tool-use-hallucinations-via.md new file mode 100644 index 0000000..06e9858 --- /dev/null +++ b/papers/items/2026-2602-08082-spectral-guardrails-for-agents-in-the-wild-detecting-tool-use-hallucinations-via.md @@ -0,0 +1,62 @@ +# Paper: Spectral Guardrails for Agents in the Wild: Detecting Tool Use Hallucinations via Attention Topology + +--- +type: paper +title: "Spectral Guardrails for Agents in the Wild: Detecting Tool Use Hallucinations via Attention Topology" +authors: Valentin Noël +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.08082 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-08 +updated_at: 2026-02-08 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - eess.SP +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.LG, cs.AI, eess.SP +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.08082 diff --git a/papers/items/2026-2602-08412-from-assistant-to-double-agent-formalizing-and-benchmarking-attacks-on-openclaw-.md b/papers/items/2026-2602-08412-from-assistant-to-double-agent-formalizing-and-benchmarking-attacks-on-openclaw-.md new file mode 100644 index 0000000..832dd57 --- /dev/null +++ b/papers/items/2026-2602-08412-from-assistant-to-double-agent-formalizing-and-benchmarking-attacks-on-openclaw-.md @@ -0,0 +1,63 @@ +# Paper: From Assistant to Double Agent: Formalizing and Benchmarking Attacks on OpenClaw for Personalized Local AI Agent + +--- +type: paper +title: "From Assistant to Double Agent: Formalizing and Benchmarking Attacks on OpenClaw for Personalized Local AI Agent" +authors: Yuhang Wang, Feiming Xu, Zheng Lin, Guangyu He, Yuzhe Huang, Haichang Gao, Zhenxing Niu, Shiguo Lian, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.08412 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-09 +updated_at: 2026-02-11 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, memory, planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.08412 diff --git a/papers/items/2026-2602-10133-agenttrace-a-structured-logging-framework-for-agent-system-observability.md b/papers/items/2026-2602-10133-agenttrace-a-structured-logging-framework-for-agent-system-observability.md new file mode 100644 index 0000000..1f86431 --- /dev/null +++ b/papers/items/2026-2602-10133-agenttrace-a-structured-logging-framework-for-agent-system-observability.md @@ -0,0 +1,62 @@ +# Paper: AgentTrace: A Structured Logging Framework for Agent System Observability + +--- +type: paper +title: "AgentTrace: A Structured Logging Framework for Agent System Observability" +authors: Adam AlSayyad, Kelvin Yuxiang Huang, Richik Pal +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.10133 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-07 +updated_at: 2026-02-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, reasoning, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.10133 diff --git a/papers/items/2026-2602-11749-air-improving-agent-safety-through-incident-response.md b/papers/items/2026-2602-11749-air-improving-agent-safety-through-incident-response.md new file mode 100644 index 0000000..5db9e2c --- /dev/null +++ b/papers/items/2026-2602-11749-air-improving-agent-safety-through-incident-response.md @@ -0,0 +1,61 @@ +# Paper: AIR: Improving Agent Safety through Incident Response + +--- +type: paper +title: "AIR: Improving Agent Safety through Incident Response" +authors: Zibo Xiao, Jun Sun, Junjie Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.11749 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-12 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.11749 diff --git a/papers/items/2026-2602-13379-unsafer-in-many-turns-benchmarking-and-defending-multi-turn-safety-risks-in-tool.md b/papers/items/2026-2602-13379-unsafer-in-many-turns-benchmarking-and-defending-multi-turn-safety-risks-in-tool.md new file mode 100644 index 0000000..8e3d005 --- /dev/null +++ b/papers/items/2026-2602-13379-unsafer-in-many-turns-benchmarking-and-defending-multi-turn-safety-risks-in-tool.md @@ -0,0 +1,65 @@ +# Paper: Unsafer in Many Turns: Benchmarking and Defending Multi-Turn Safety Risks in Tool-Using Agents + +--- +type: paper +title: "Unsafer in Many Turns: Benchmarking and Defending Multi-Turn Safety Risks in Tool-Using Agents" +authors: Xu Li, Simon Yu, Minzhou Pan, Yiyou Sun, Bo Li, Dawn Song, Xue Lin, Weiyan Shi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.13379 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-13 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.CL + - cs.LG + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.CR, cs.AI, cs.CL, cs.LG, cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.13379 diff --git a/papers/items/2026-2602-13530-remem-reasoning-with-episodic-memory-in-language-agent.md b/papers/items/2026-2602-13530-remem-reasoning-with-episodic-memory-in-language-agent.md new file mode 100644 index 0000000..0aa8f59 --- /dev/null +++ b/papers/items/2026-2602-13530-remem-reasoning-with-episodic-memory-in-language-agent.md @@ -0,0 +1,62 @@ +# Paper: REMem: Reasoning with Episodic Memory in Language Agent + +--- +type: paper +title: "REMem: Reasoning with Episodic Memory in Language Agent" +authors: Yiheng Shu, Saisri Padmaja Jonnalagedda, Xiang Gao, Bernal Jiménez Gutiérrez, Weijian Qi, Kamalika Das, Huan Sun, Yu Su +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.13530 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-13 +updated_at: 2026-02-28 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.13530 diff --git a/papers/items/2026-2602-13665-hyfunc-accelerating-llm-based-function-calls-for-agentic-ai-through-hybrid-model.md b/papers/items/2026-2602-13665-hyfunc-accelerating-llm-based-function-calls-for-agentic-ai-through-hybrid-model.md new file mode 100644 index 0000000..07d6fd3 --- /dev/null +++ b/papers/items/2026-2602-13665-hyfunc-accelerating-llm-based-function-calls-for-agentic-ai-through-hybrid-model.md @@ -0,0 +1,59 @@ +# Paper: HyFunc: Accelerating LLM-based Function Calls for Agentic AI through Hybrid-Model Cascade and Dynamic Templating + +--- +type: paper +title: "HyFunc: Accelerating LLM-based Function Calls for Agentic AI through Hybrid-Model Cascade and Dynamic Templating" +authors: Weibin Liao, Jian-guang Lou, Haoyi Xiong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.13665 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-14 +updated_at: 2026-02-14 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.13665 diff --git a/papers/items/2026-2602-14234-redsearcher-a-scalable-and-cost-efficient-framework-for-long-horizon-search-agen.md b/papers/items/2026-2602-14234-redsearcher-a-scalable-and-cost-efficient-framework-for-long-horizon-search-agen.md new file mode 100644 index 0000000..61db05b --- /dev/null +++ b/papers/items/2026-2602-14234-redsearcher-a-scalable-and-cost-efficient-framework-for-long-horizon-search-agen.md @@ -0,0 +1,62 @@ +# Paper: REDSearcher: A Scalable and Cost-Efficient Framework for Long-Horizon Search Agents + +--- +type: paper +title: "REDSearcher: A Scalable and Cost-Efficient Framework for Long-Horizon Search Agents" +authors: Zheng Chu, Xiao Wang, Jack Hong, Huiming Fan, Yuqi Huang, Yue Yang, Guohai Xu, Chenxiao Zhao, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.14234 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-15 +updated_at: 2026-02-15 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, planning, rag, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.14234 diff --git a/papers/items/2026-2602-14281-mcpshield-a-security-cognition-layer-for-adaptive-trust-calibration-in-model-con.md b/papers/items/2026-2602-14281-mcpshield-a-security-cognition-layer-for-adaptive-trust-calibration-in-model-con.md new file mode 100644 index 0000000..061d77e --- /dev/null +++ b/papers/items/2026-2602-14281-mcpshield-a-security-cognition-layer-for-adaptive-trust-calibration-in-model-con.md @@ -0,0 +1,62 @@ +# Paper: MCPShield: A Security Cognition Layer for Adaptive Trust Calibration in Model Context Protocol Agents + +--- +type: paper +title: "MCPShield: A Security Cognition Layer for Adaptive Trust Calibration in Model Context Protocol Agents" +authors: Zhenhong Zhou, Yuanhe Zhang, Hongwei Cai, Moayad Aloqaily, Ouns Bouachir, Linsey Pang, Prakhar Mehrotra, Kun Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.14281 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-15 +updated_at: 2026-02-24 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, computer-use, reasoning, tool-use +- arXiv categories: cs.CR, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.14281 diff --git a/papers/items/2026-2602-16931-narrow-fine-tuning-erodes-safety-alignment-in-vision-language-agents.md b/papers/items/2026-2602-16931-narrow-fine-tuning-erodes-safety-alignment-in-vision-language-agents.md new file mode 100644 index 0000000..187691d --- /dev/null +++ b/papers/items/2026-2602-16931-narrow-fine-tuning-erodes-safety-alignment-in-vision-language-agents.md @@ -0,0 +1,59 @@ +# Paper: Narrow Fine-Tuning Erodes Safety Alignment in Vision-Language Agents + +--- +type: paper +title: Narrow Fine-Tuning Erodes Safety Alignment in Vision-Language Agents +authors: Idhant Gulati, Shivam Raval +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.16931 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-18 +updated_at: 2026-03-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, agent-safety +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.16931 diff --git a/papers/items/2026-2602-18456-beyond-single-channel-agentic-benchmarking.md b/papers/items/2026-2602-18456-beyond-single-channel-agentic-benchmarking.md new file mode 100644 index 0000000..4793625 --- /dev/null +++ b/papers/items/2026-2602-18456-beyond-single-channel-agentic-benchmarking.md @@ -0,0 +1,61 @@ +# Paper: Beyond single-channel agentic benchmarking + +--- +type: paper +title: Beyond single-channel agentic benchmarking +authors: Nelu D. Radpour +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.18456 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-05 +updated_at: 2026-02-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY + - cs.AI + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety +- arXiv categories: cs.CY, cs.AI, cs.HC +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.18456 diff --git a/papers/items/2026-2602-19008-capable-but-unreliable-canonical-path-deviation-as-a-causal-mechanism-of-agent-f.md b/papers/items/2026-2602-19008-capable-but-unreliable-canonical-path-deviation-as-a-causal-mechanism-of-agent-f.md new file mode 100644 index 0000000..4a7311b --- /dev/null +++ b/papers/items/2026-2602-19008-capable-but-unreliable-canonical-path-deviation-as-a-causal-mechanism-of-agent-f.md @@ -0,0 +1,62 @@ +# Paper: Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks + +--- +type: paper +title: "Capable but Unreliable: Canonical Path Deviation as a Causal Mechanism of Agent Failure in Long-Horizon Tasks" +authors: Wilson Y. Lee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.19008 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-22 +updated_at: 2026-02-22 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use +- arXiv categories: cs.CL, cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.19008 diff --git a/papers/items/2026-2602-21127-are-you-sure-an-empirical-study-of-human-perception-vulnerability-in-llm-driven-.md b/papers/items/2026-2602-21127-are-you-sure-an-empirical-study-of-human-perception-vulnerability-in-llm-driven-.md new file mode 100644 index 0000000..4b7032b --- /dev/null +++ b/papers/items/2026-2602-21127-are-you-sure-an-empirical-study-of-human-perception-vulnerability-in-llm-driven-.md @@ -0,0 +1,63 @@ +# Paper: "Are You Sure?": An Empirical Study of Human Perception Vulnerability in LLM-Driven Agentic Systems + +--- +type: paper +title: "\"Are You Sure?\": An Empirical Study of Human Perception Vulnerability in LLM-Driven Agentic Systems" +authors: Xinfeng Li, Shenyu Dai, Kelong Zheng, Yue Xiao, Gelei Deng, Wei Dong, Xiaofeng Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.21127 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-24 +updated_at: 2026-02-24 +status: queued +relevance: high +topics: + - agent-safety + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.HC + - cs.AI + - cs.CR + - cs.SI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, tool-use, workflow-agent +- arXiv categories: cs.HC, cs.AI, cs.CR, cs.SI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.21127 diff --git a/papers/items/2026-2602-23320-parammem-augmenting-language-agents-with-parametric-reflective-memory.md b/papers/items/2026-2602-23320-parammem-augmenting-language-agents-with-parametric-reflective-memory.md new file mode 100644 index 0000000..da91dca --- /dev/null +++ b/papers/items/2026-2602-23320-parammem-augmenting-language-agents-with-parametric-reflective-memory.md @@ -0,0 +1,61 @@ +# Paper: ParamMem: Augmenting Language Agents with Parametric Reflective Memory + +--- +type: paper +title: "ParamMem: Augmenting Language Agents with Parametric Reflective Memory" +authors: Tianjun Yao, Yongqiang Chen, Yujia Zheng, Pan Li, Zhiqiang Shen, Kun Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2602.23320 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-26 +updated_at: 2026-02-27 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, reasoning +- arXiv categories: cs.LG, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2602.23320 diff --git a/papers/items/2026-2603-00131-thought-virus-viral-misalignment-via-subliminal-prompting-in-multi-agent-systems.md b/papers/items/2026-2603-00131-thought-virus-viral-misalignment-via-subliminal-prompting-in-multi-agent-systems.md new file mode 100644 index 0000000..9e17230 --- /dev/null +++ b/papers/items/2026-2603-00131-thought-virus-viral-misalignment-via-subliminal-prompting-in-multi-agent-systems.md @@ -0,0 +1,61 @@ +# Paper: Thought Virus: Viral Misalignment via Subliminal Prompting in Multi-Agent Systems + +--- +type: paper +title: "Thought Virus: Viral Misalignment via Subliminal Prompting in Multi-Agent Systems" +authors: Moritz Weckbecker, Jonas Müller, Ben Hagag, Michael Mulet +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.00131 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-23 +updated_at: 2026-02-23 +status: queued +relevance: high +topics: + - agent-safety + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, multi-agent, tool-use +- arXiv categories: cs.MA, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.00131 diff --git a/papers/items/2026-2603-00623-tracesir-a-multi-agent-framework-for-structured-analysis-and-reporting-of-agenti.md b/papers/items/2026-2603-00623-tracesir-a-multi-agent-framework-for-structured-analysis-and-reporting-of-agenti.md new file mode 100644 index 0000000..1053dff --- /dev/null +++ b/papers/items/2026-2603-00623-tracesir-a-multi-agent-framework-for-structured-analysis-and-reporting-of-agenti.md @@ -0,0 +1,63 @@ +# Paper: TraceSIR: A Multi-Agent Framework for Structured Analysis and Reporting of Agentic Execution Traces + +--- +type: paper +title: "TraceSIR: A Multi-Agent Framework for Structured Analysis and Reporting of Agentic Execution Traces" +authors: Shu-Xun Yang, Cunxiang Wang, Haoke Zhang, Wenbo Yu, Lindong Wu, Jiayi Gui, Dayong Yang, Yukuo Cen, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.00623 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-28 +updated_at: 2026-02-28 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, coding-agent, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.00623 diff --git a/papers/items/2026-2603-00801-the-synthetic-web-adversarially-curated-mini-internets-for-diagnosing-epistemic-.md b/papers/items/2026-2603-00801-the-synthetic-web-adversarially-curated-mini-internets-for-diagnosing-epistemic-.md new file mode 100644 index 0000000..9e8b277 --- /dev/null +++ b/papers/items/2026-2603-00801-the-synthetic-web-adversarially-curated-mini-internets-for-diagnosing-epistemic-.md @@ -0,0 +1,62 @@ +# Paper: The Synthetic Web: Adversarially-Curated Mini-Internets for Diagnosing Epistemic Weaknesses of Language Agents + +--- +type: paper +title: "The Synthetic Web: Adversarially-Curated Mini-Internets for Diagnosing Epistemic Weaknesses of Language Agents" +authors: Shrey Shah, Levent Ozgur +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.00801 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-28 +updated_at: 2026-02-28 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, embodied-agent, rag, tool-use +- arXiv categories: cs.AI, cs.IR +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.00801 diff --git a/papers/items/2026-2603-01438-enhancing-persona-following-at-decoding-time-via-dynamic-importance-estimation-f.md b/papers/items/2026-2603-01438-enhancing-persona-following-at-decoding-time-via-dynamic-importance-estimation-f.md new file mode 100644 index 0000000..37e83ca --- /dev/null +++ b/papers/items/2026-2603-01438-enhancing-persona-following-at-decoding-time-via-dynamic-importance-estimation-f.md @@ -0,0 +1,64 @@ +# Paper: Enhancing Persona Following at Decoding Time via Dynamic Importance Estimation for Role-Playing Agents + +--- +type: paper +title: Enhancing Persona Following at Decoding Time via Dynamic Importance Estimation for Role-Playing Agents +authors: Yuxin Liu, Mingye Zhu, Siyuan Liu, Bo Hu, Lei Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.01438 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-02 +updated_at: 2026-03-02 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - computer-use + - planning + - rag + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-safety, coding-agent, computer-use, planning, rag, world-model +- arXiv categories: cs.CL, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.01438 diff --git a/papers/items/2026-2603-01712-ft-dojo-towards-autonomous-llm-fine-tuning-with-language-agents.md b/papers/items/2026-2603-01712-ft-dojo-towards-autonomous-llm-fine-tuning-with-language-agents.md new file mode 100644 index 0000000..fbd15c2 --- /dev/null +++ b/papers/items/2026-2603-01712-ft-dojo-towards-autonomous-llm-fine-tuning-with-language-agents.md @@ -0,0 +1,61 @@ +# Paper: FT-Dojo: Towards Autonomous LLM Fine-Tuning with Language Agents + +--- +type: paper +title: "FT-Dojo: Towards Autonomous LLM Fine-Tuning with Language Agents" +authors: Qizheng Li, Yifei Zhang, Xiao Yang, Xu Yang, Zhuo Wang, Weiqing Liu, Jiang Bian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.01712 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-02 +updated_at: 2026-05-20 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, coding-agent, planning +- arXiv categories: cs.AI, cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.01712 diff --git a/papers/items/2026-2603-02711-a-natural-language-agentic-approach-to-study-affective-polarization.md b/papers/items/2026-2603-02711-a-natural-language-agentic-approach-to-study-affective-polarization.md new file mode 100644 index 0000000..e02a70e --- /dev/null +++ b/papers/items/2026-2603-02711-a-natural-language-agentic-approach-to-study-affective-polarization.md @@ -0,0 +1,60 @@ +# Paper: A Natural Language Agentic Approach to Study Affective Polarization + +--- +type: paper +title: A Natural Language Agentic Approach to Study Affective Polarization +authors: Stephanie Anneris Malvicini, Ewelina Gajewska, Arda Derbent, Katarzyna Budzynska, Jarosław A. Chudziak, Maria Vanina Martinez +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.02711 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-03 +updated_at: 2026-03-03 +status: queued +relevance: high +topics: + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: multi-agent, rag, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.02711 diff --git a/papers/items/2026-2603-03515-the-controllability-trap-a-governance-framework-for-military-ai-agents.md b/papers/items/2026-2603-03515-the-controllability-trap-a-governance-framework-for-military-ai-agents.md new file mode 100644 index 0000000..b5e6d92 --- /dev/null +++ b/papers/items/2026-2603-03515-the-controllability-trap-a-governance-framework-for-military-ai-agents.md @@ -0,0 +1,63 @@ +# Paper: The Controllability Trap: A Governance Framework for Military AI Agents + +--- +type: paper +title: "The Controllability Trap: A Governance Framework for Military AI Agents" +authors: Subramanyam Sahoo +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.03515 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-03 +updated_at: 2026-03-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, planning, tool-use, world-model +- arXiv categories: cs.CY, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.03515 diff --git a/papers/items/2026-2603-03680-mage-meta-reinforcement-learning-for-language-agents-toward-strategic-exploratio.md b/papers/items/2026-2603-03680-mage-meta-reinforcement-learning-for-language-agents-toward-strategic-exploratio.md new file mode 100644 index 0000000..9153295 --- /dev/null +++ b/papers/items/2026-2603-03680-mage-meta-reinforcement-learning-for-language-agents-toward-strategic-exploratio.md @@ -0,0 +1,61 @@ +# Paper: MAGE: Meta-Reinforcement Learning for Language Agents toward Strategic Exploration and Exploitation + +--- +type: paper +title: "MAGE: Meta-Reinforcement Learning for Language Agents toward Strategic Exploration and Exploitation" +authors: Lu Yang, Zelai Xu, Minyang Xie, Jiaxuan Gao, Zhao Shok, Yu Wang, Yi Wu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.03680 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-04 +updated_at: 2026-03-04 +status: queued +relevance: high +topics: + - memory + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: memory, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.03680 diff --git a/papers/items/2026-2603-05553-eigendata-a-self-evolving-multi-agent-platform-for-function-calling-data-synthes.md b/papers/items/2026-2603-05553-eigendata-a-self-evolving-multi-agent-platform-for-function-calling-data-synthes.md new file mode 100644 index 0000000..b6dc370 --- /dev/null +++ b/papers/items/2026-2603-05553-eigendata-a-self-evolving-multi-agent-platform-for-function-calling-data-synthes.md @@ -0,0 +1,63 @@ +# Paper: EigenData: A Self-Evolving Multi-Agent Platform for Function-Calling Data Synthesis, Auditing, and Repair + +--- +type: paper +title: "EigenData: A Self-Evolving Multi-Agent Platform for Function-Calling Data Synthesis, Auditing, and Repair" +authors: Jiaao Chen, Jingyuan Qi, Mingye Gao, Wei-Chen Wang, Hanrui Wang, Di Jin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.05553 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-05 +updated_at: 2026-03-05 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, coding-agent, multi-agent, tool-use +- arXiv categories: cs.SE, cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.05553 diff --git a/papers/items/2026-2603-05578-tool-genesis-a-task-driven-tool-creation-benchmark-for-self-evolving-language-ag.md b/papers/items/2026-2603-05578-tool-genesis-a-task-driven-tool-creation-benchmark-for-self-evolving-language-ag.md new file mode 100644 index 0000000..6576f18 --- /dev/null +++ b/papers/items/2026-2603-05578-tool-genesis-a-task-driven-tool-creation-benchmark-for-self-evolving-language-ag.md @@ -0,0 +1,61 @@ +# Paper: Tool-Genesis: A Task-Driven Tool Creation Benchmark for Self-Evolving Language Agent + +--- +type: paper +title: "Tool-Genesis: A Task-Driven Tool Creation Benchmark for Self-Evolving Language Agent" +authors: Bowei Xia, Mengkang Hu, Shijian Wang, Jiarui Jin, Wenxiang Jiao, Yuan Lu, Kexin Li, Ping Luo +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.05578 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-05 +updated_at: 2026-03-05 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, computer-use, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.05578 diff --git a/papers/items/2026-2603-07496-from-thinker-to-society-security-in-hierarchical-autonomy-evolution-of-ai-agents.md b/papers/items/2026-2603-07496-from-thinker-to-society-security-in-hierarchical-autonomy-evolution-of-ai-agents.md new file mode 100644 index 0000000..62d0fc8 --- /dev/null +++ b/papers/items/2026-2603-07496-from-thinker-to-society-security-in-hierarchical-autonomy-evolution-of-ai-agents.md @@ -0,0 +1,64 @@ +# Paper: From Thinker to Society: Security in Hierarchical Autonomy Evolution of AI Agents + +--- +type: paper +title: "From Thinker to Society: Security in Hierarchical Autonomy Evolution of AI Agents" +authors: Xiaolei Zhang, Lu Zhou, Xiaogang Xu, Jiafei Wu, Tianyu Du, Heqing Huang, Hao Peng, Zhe Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.07496 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-08 +updated_at: 2026-03-21 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, reasoning, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.07496 diff --git a/papers/items/2026-2603-07557-agentraft-automated-detection-of-data-over-exposure-in-llm-agents.md b/papers/items/2026-2603-07557-agentraft-automated-detection-of-data-over-exposure-in-llm-agents.md new file mode 100644 index 0000000..20fcdbd --- /dev/null +++ b/papers/items/2026-2603-07557-agentraft-automated-detection-of-data-over-exposure-in-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: AgentRaft: Automated Detection of Data Over-Exposure in LLM Agents + +--- +type: paper +title: "AgentRaft: Automated Detection of Data Over-Exposure in LLM Agents" +authors: Yixi Lin, Jiangrong Wu, Yuhong Nan, Xueqiang Wang, Xinyuan Zhang, Zibin Zheng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.07557 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-08 +updated_at: 2026-03-08 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, rag, reasoning, tool-use +- arXiv categories: cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.07557 diff --git a/papers/items/2026-2603-07980-onemillion-bench-how-far-are-language-agents-from-human-experts.md b/papers/items/2026-2603-07980-onemillion-bench-how-far-are-language-agents-from-human-experts.md new file mode 100644 index 0000000..8667452 --- /dev/null +++ b/papers/items/2026-2603-07980-onemillion-bench-how-far-are-language-agents-from-human-experts.md @@ -0,0 +1,63 @@ +# Paper: \$OneMillion-Bench: How Far are Language Agents from Human Experts? + +--- +type: paper +title: "\\$OneMillion-Bench: How Far are Language Agents from Human Experts?" +authors: Qianyu Yang, Yang Liu, Jiaqi Li, Jun Bai, Hao Chen, Kaiyuan Chen, Tiliang Duan, Jiayun Dong, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.07980 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-09 +updated_at: 2026-03-09 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, planning, reasoning, tool-use +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.07980 diff --git a/papers/items/2026-2603-08721-kernelcraft-benchmarking-for-agentic-close-to-metal-kernel-generation-on-emergin.md b/papers/items/2026-2603-08721-kernelcraft-benchmarking-for-agentic-close-to-metal-kernel-generation-on-emergin.md new file mode 100644 index 0000000..260c6d4 --- /dev/null +++ b/papers/items/2026-2603-08721-kernelcraft-benchmarking-for-agentic-close-to-metal-kernel-generation-on-emergin.md @@ -0,0 +1,62 @@ +# Paper: KernelCraft: Benchmarking for Agentic Close-to-Metal Kernel Generation on Emerging Hardware + +--- +type: paper +title: "KernelCraft: Benchmarking for Agentic Close-to-Metal Kernel Generation on Emerging Hardware" +authors: Jiayi Nie, Haoran Wu, Yao Lai, Zeyu Cao, Cheng Zhang, Binglei Lou, Erwei Wang, Jianyi Cheng, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.08721 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-10 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AR + - cs.LG + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, reasoning, workflow-agent +- arXiv categories: cs.AR, cs.LG, cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.08721 diff --git a/papers/items/2026-2603-09002-security-considerations-for-multi-agent-systems.md b/papers/items/2026-2603-09002-security-considerations-for-multi-agent-systems.md new file mode 100644 index 0000000..9fa5b26 --- /dev/null +++ b/papers/items/2026-2603-09002-security-considerations-for-multi-agent-systems.md @@ -0,0 +1,66 @@ +# Paper: Security Considerations for Multi-agent Systems + +--- +type: paper +title: Security Considerations for Multi-agent Systems +authors: Tam Nguyen, Moses Ndebugre, Dheeraj Arremsetty +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.09002 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-09 +updated_at: 2026-04-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - multi-agent + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, multi-agent, planning, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.09002 diff --git a/papers/items/2026-2603-10492-human-ai-co-reasoning-for-clinical-diagnosis-with-evidence-integrated-language-a.md b/papers/items/2026-2603-10492-human-ai-co-reasoning-for-clinical-diagnosis-with-evidence-integrated-language-a.md new file mode 100644 index 0000000..5125a9c --- /dev/null +++ b/papers/items/2026-2603-10492-human-ai-co-reasoning-for-clinical-diagnosis-with-evidence-integrated-language-a.md @@ -0,0 +1,63 @@ +# Paper: Human-AI Co-reasoning for Clinical Diagnosis with Evidence-Integrated Language Agent + +--- +type: paper +title: Human-AI Co-reasoning for Clinical Diagnosis with Evidence-Integrated Language Agent +authors: Zhongzhen Huang, Yan Ling, Hong Chen, Ye Feng, Li Wu, Linjie Mu, Shaoting Zhang, Xiaofan Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.10492 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-11 +updated_at: 2026-03-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, reasoning, workflow-agent +- arXiv categories: cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.10492 diff --git a/papers/items/2026-2603-11088-the-attack-and-defense-landscape-of-agentic-ai-a-comprehensive-survey.md b/papers/items/2026-2603-11088-the-attack-and-defense-landscape-of-agentic-ai-a-comprehensive-survey.md new file mode 100644 index 0000000..632802f --- /dev/null +++ b/papers/items/2026-2603-11088-the-attack-and-defense-landscape-of-agentic-ai-a-comprehensive-survey.md @@ -0,0 +1,61 @@ +# Paper: The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey + +--- +type: paper +title: "The Attack and Defense Landscape of Agentic AI: A Comprehensive Survey" +authors: Juhee Kim, Xiaoyuan Liu, Zhun Wang, Shi Qiu, Bo Li, Wenbo Guo, Dawn Song +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.11088 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-11 +updated_at: 2026-03-11 +status: queued +relevance: high +topics: + - agent-safety + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.11088 diff --git a/papers/items/2026-2603-11890-quare-quality-aware-requirements-analysis-through-multi-agent-dialectical-negoti.md b/papers/items/2026-2603-11890-quare-quality-aware-requirements-analysis-through-multi-agent-dialectical-negoti.md new file mode 100644 index 0000000..b1f8213 --- /dev/null +++ b/papers/items/2026-2603-11890-quare-quality-aware-requirements-analysis-through-multi-agent-dialectical-negoti.md @@ -0,0 +1,61 @@ +# Paper: QUARE: Quality-Aware Requirements Analysis through Multi-Agent Dialectical Negotiation + +--- +type: paper +title: "QUARE: Quality-Aware Requirements Analysis through Multi-Agent Dialectical Negotiation" +authors: Haowei Cheng, Milhan Kim, Foutse Khomh, Teeradaj Racharak, Nobukazu Yoshioka, Naoyasu Ubayashi, Hironori Washizaki +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.11890 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-12 +updated_at: 2026-06-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag +- arXiv categories: cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.11890 diff --git a/papers/items/2026-2603-15309-cctu-a-benchmark-for-tool-use-under-complex-constraints.md b/papers/items/2026-2603-15309-cctu-a-benchmark-for-tool-use-under-complex-constraints.md new file mode 100644 index 0000000..291cbfc --- /dev/null +++ b/papers/items/2026-2603-15309-cctu-a-benchmark-for-tool-use-under-complex-constraints.md @@ -0,0 +1,61 @@ +# Paper: CCTU: A Benchmark for Tool Use under Complex Constraints + +--- +type: paper +title: "CCTU: A Benchmark for Tool Use under Complex Constraints" +authors: Junjie Ye, Guoqiang Zhang, Wenjie Fu, Tao Gui, Qi Zhang, Xuanjing Huang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.15309 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-16 +updated_at: 2026-03-16 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, rag, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.15309 diff --git a/papers/items/2026-2603-15666-compiled-memory-not-more-information-but-more-precise-instructions-for-language-.md b/papers/items/2026-2603-15666-compiled-memory-not-more-information-but-more-precise-instructions-for-language-.md new file mode 100644 index 0000000..00361f0 --- /dev/null +++ b/papers/items/2026-2603-15666-compiled-memory-not-more-information-but-more-precise-instructions-for-language-.md @@ -0,0 +1,59 @@ +# Paper: Compiled Memory: Not More Information, but More Precise Instructions for Language Agents + +--- +type: paper +title: "Compiled Memory: Not More Information, but More Precise Instructions for Language Agents" +authors: James Rhodes, George Kang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.15666 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-12 +updated_at: 2026-03-12 +status: queued +relevance: high +topics: + - memory + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: memory, rag +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.15666 diff --git a/papers/items/2026-2603-16734-differential-harm-propensity-in-personalized-llm-agents-the-curious-case-of-ment.md b/papers/items/2026-2603-16734-differential-harm-propensity-in-personalized-llm-agents-the-curious-case-of-ment.md new file mode 100644 index 0000000..df68fde --- /dev/null +++ b/papers/items/2026-2603-16734-differential-harm-propensity-in-personalized-llm-agents-the-curious-case-of-ment.md @@ -0,0 +1,62 @@ +# Paper: Differential Harm Propensity in Personalized LLM Agents: The Curious Case of Mental Health Disclosure + +--- +type: paper +title: "Differential Harm Propensity in Personalized LLM Agents: The Curious Case of Mental Health Disclosure" +authors: Caglar Yildirim +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.16734 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-17 +updated_at: 2026-03-17 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.16734 diff --git a/papers/items/2026-2603-17392-agentic-cognitive-profiling-realigning-automated-alzheimer-s-disease-detection-w.md b/papers/items/2026-2603-17392-agentic-cognitive-profiling-realigning-automated-alzheimer-s-disease-detection-w.md new file mode 100644 index 0000000..cc6d9bf --- /dev/null +++ b/papers/items/2026-2603-17392-agentic-cognitive-profiling-realigning-automated-alzheimer-s-disease-detection-w.md @@ -0,0 +1,61 @@ +# Paper: Agentic Cognitive Profiling: Realigning Automated Alzheimer's Disease Detection with Clinical Construct Validity + +--- +type: paper +title: "Agentic Cognitive Profiling: Realigning Automated Alzheimer's Disease Detection with Clinical Construct Validity" +authors: Jiawen Kang, Kun Li, Dongrui Han, Jinchao Li, Junan Li, Lingwei Meng, Xixin Wu, Helen Meng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.17392 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-18 +updated_at: 2026-03-18 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.IR + - q-bio.NC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.MA, cs.IR, q-bio.NC +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.17392 diff --git a/papers/items/2026-2603-18245-who-tests-the-testers-systematic-enumeration-and-coverage-audit-of-llm-agent-too.md b/papers/items/2026-2603-18245-who-tests-the-testers-systematic-enumeration-and-coverage-audit-of-llm-agent-too.md new file mode 100644 index 0000000..4afb96e --- /dev/null +++ b/papers/items/2026-2603-18245-who-tests-the-testers-systematic-enumeration-and-coverage-audit-of-llm-agent-too.md @@ -0,0 +1,63 @@ +# Paper: Who Tests the Testers? Systematic Enumeration and Coverage Audit of LLM Agent Tool Call Safety + +--- +type: paper +title: Who Tests the Testers? Systematic Enumeration and Coverage Audit of LLM Agent Tool Call Safety +authors: Xuan Chen, Lu Yan, Ruqi Zhang, Xiangyu Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.18245 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-18 +updated_at: 2026-03-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, rag, tool-use, workflow-agent +- arXiv categories: cs.SE, cs.CR +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.18245 diff --git a/papers/items/2026-2603-19469-a-framework-for-formalizing-llm-agent-security.md b/papers/items/2026-2603-19469-a-framework-for-formalizing-llm-agent-security.md new file mode 100644 index 0000000..4d96660 --- /dev/null +++ b/papers/items/2026-2603-19469-a-framework-for-formalizing-llm-agent-security.md @@ -0,0 +1,61 @@ +# Paper: A Framework for Formalizing LLM Agent Security + +--- +type: paper +title: A Framework for Formalizing LLM Agent Security +authors: Vincent Siu, Jingxuan He, Kyle Montgomery, Zhun Wang, Neil Gong, Chenguang Wang, Dawn Song +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.19469 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-19 +updated_at: 2026-03-19 +status: queued +relevance: high +topics: + - agent-safety + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, memory, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.19469 diff --git a/papers/items/2026-2603-19684-tsegagent-zero-shot-tooth-segmentation-via-geometry-aware-vision-language-agents.md b/papers/items/2026-2603-19684-tsegagent-zero-shot-tooth-segmentation-via-geometry-aware-vision-language-agents.md new file mode 100644 index 0000000..4d46516 --- /dev/null +++ b/papers/items/2026-2603-19684-tsegagent-zero-shot-tooth-segmentation-via-geometry-aware-vision-language-agents.md @@ -0,0 +1,62 @@ +# Paper: TSegAgent: Zero-Shot Tooth Segmentation via Geometry-Aware Vision-Language Agents + +--- +type: paper +title: "TSegAgent: Zero-Shot Tooth Segmentation via Geometry-Aware Vision-Language Agents" +authors: Shaojie Zhuang, Lu Yin, Guangshun Wei, Yunpeng Li, Xilu Wang, Yuanfeng Zhou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.19684 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-20 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, coding-agent, rag, reasoning, tool-use +- arXiv categories: cs.CV +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.19684 diff --git a/papers/items/2026-2603-21357-agenther-hindsight-experience-replay-for-llm-agent-trajectory-relabeling.md b/papers/items/2026-2603-21357-agenther-hindsight-experience-replay-for-llm-agent-trajectory-relabeling.md new file mode 100644 index 0000000..0db7643 --- /dev/null +++ b/papers/items/2026-2603-21357-agenther-hindsight-experience-replay-for-llm-agent-trajectory-relabeling.md @@ -0,0 +1,62 @@ +# Paper: AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling + +--- +type: paper +title: "AgentHER: Hindsight Experience Replay for LLM Agent Trajectory Relabeling" +authors: Liang Ding +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.21357 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-22 +updated_at: 2026-05-10 +status: queued +relevance: high +topics: + - computer-use + - memory + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: computer-use, memory, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.21357 diff --git a/papers/items/2026-2603-21564-toward-a-theory-of-hierarchical-memory-for-language-agents.md b/papers/items/2026-2603-21564-toward-a-theory-of-hierarchical-memory-for-language-agents.md new file mode 100644 index 0000000..2c82af7 --- /dev/null +++ b/papers/items/2026-2603-21564-toward-a-theory-of-hierarchical-memory-for-language-agents.md @@ -0,0 +1,64 @@ +# Paper: Toward a Theory of Hierarchical Memory for Language Agents + +--- +type: paper +title: Toward a Theory of Hierarchical Memory for Language Agents +authors: Yashar Talebirad, Ali Parsaee, Csongor Y. Szepesvari, Amirhossein Nadiri, Osmar Zaiane +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.21564 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-23 +updated_at: 2026-03-23 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI + - cs.IT + - cs.SI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.IR, cs.AI, cs.IT, cs.SI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.21564 diff --git a/papers/items/2026-2603-24257-memory-augmented-vision-language-agents-for-persistent-and-semantically-consiste.md b/papers/items/2026-2603-24257-memory-augmented-vision-language-agents-for-persistent-and-semantically-consiste.md new file mode 100644 index 0000000..5dbe041 --- /dev/null +++ b/papers/items/2026-2603-24257-memory-augmented-vision-language-agents-for-persistent-and-semantically-consiste.md @@ -0,0 +1,60 @@ +# Paper: Memory-Augmented Vision-Language Agents for Persistent and Semantically Consistent Object Captioning + +--- +type: paper +title: Memory-Augmented Vision-Language Agents for Persistent and Semantically Consistent Object Captioning +authors: Tommaso Galliena, Stefano Rosa, Tommaso Apicella, Pietro Morerio, Alessio Del Bue, Lorenzo Natale +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.24257 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-25 +updated_at: 2026-03-30 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, embodied-agent, memory +- arXiv categories: cs.CV +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.24257 diff --git a/papers/items/2026-2603-25353-safeguard-asf-sr-agentic-humanoid-robot-system-for-autonomous-industrial-safety.md b/papers/items/2026-2603-25353-safeguard-asf-sr-agentic-humanoid-robot-system-for-autonomous-industrial-safety.md new file mode 100644 index 0000000..a3c1dcd --- /dev/null +++ b/papers/items/2026-2603-25353-safeguard-asf-sr-agentic-humanoid-robot-system-for-autonomous-industrial-safety.md @@ -0,0 +1,62 @@ +# Paper: SafeGuard ASF: SR Agentic Humanoid Robot System for Autonomous Industrial Safety + +--- +type: paper +title: "SafeGuard ASF: SR Agentic Humanoid Robot System for Autonomous Industrial Safety" +authors: Thanh Nguyen Canh, Thang Tran Viet, Thanh Tuan Tran, Ben Wei Lim +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.25353 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-26 +updated_at: 2026-03-26 +status: queued +relevance: high +topics: + - agent-safety + - embodied-agent + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, embodied-agent, reasoning, tool-use, world-model +- arXiv categories: cs.RO +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.25353 diff --git a/papers/items/2026-2603-27148-safetydrift-predicting-when-ai-agents-cross-the-line-before-they-actually-do.md b/papers/items/2026-2603-27148-safetydrift-predicting-when-ai-agents-cross-the-line-before-they-actually-do.md new file mode 100644 index 0000000..8595609 --- /dev/null +++ b/papers/items/2026-2603-27148-safetydrift-predicting-when-ai-agents-cross-the-line-before-they-actually-do.md @@ -0,0 +1,60 @@ +# Paper: SafetyDrift: Predicting When AI Agents Cross the Line Before They Actually Do + +--- +type: paper +title: "SafetyDrift: Predicting When AI Agents Cross the Line Before They Actually Do" +authors: Aditya Dhodapkar, Farhaan Pishori +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.27148 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-28 +updated_at: 2026-03-28 +status: queued +relevance: high +topics: + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.27148 diff --git a/papers/items/2026-2603-28166-evaluating-privilege-usage-of-agents-with-real-world-tools.md b/papers/items/2026-2603-28166-evaluating-privilege-usage-of-agents-with-real-world-tools.md new file mode 100644 index 0000000..00d36c1 --- /dev/null +++ b/papers/items/2026-2603-28166-evaluating-privilege-usage-of-agents-with-real-world-tools.md @@ -0,0 +1,63 @@ +# Paper: Evaluating Privilege Usage of Agents with Real-World Tools + +--- +type: paper +title: Evaluating Privilege Usage of Agents with Real-World Tools +authors: Quan Zhang, Lianhang Fu, Lvsi Lian, Gwihwan Go, Yujue Wang, Chijin Zhou, Yu Jiang, Geguang Pu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.28166 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-30 +updated_at: 2026-04-20 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, rag, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.28166 diff --git a/papers/items/2026-2603-28428-synergy-a-next-generation-general-purpose-agent-for-open-agentic-web.md b/papers/items/2026-2603-28428-synergy-a-next-generation-general-purpose-agent-for-open-agentic-web.md new file mode 100644 index 0000000..e0cee35 --- /dev/null +++ b/papers/items/2026-2603-28428-synergy-a-next-generation-general-purpose-agent-for-open-agentic-web.md @@ -0,0 +1,64 @@ +# Paper: Synergy: A Next-Generation General-Purpose Agent for Open Agentic Web + +--- +type: paper +title: "Synergy: A Next-Generation General-Purpose Agent for Open Agentic Web" +authors: Xiaohang Nie, Zihan Guo, Kezhuo Yang, Zhichong Zheng, Bochen Ge, Shuai Pan, Zeyi Chen, Youling Xiang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.28428 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-30 +updated_at: 2026-03-30 +status: queued +relevance: high +topics: + - coding-agent + - embodied-agent + - memory + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: coding-agent, embodied-agent, memory, multi-agent, rag, tool-use +- arXiv categories: cs.CY, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.28428 diff --git a/papers/items/2026-2603-28900-robust-multi-agent-reinforcement-learning-for-small-uas-separation-assurance-und.md b/papers/items/2026-2603-28900-robust-multi-agent-reinforcement-learning-for-small-uas-separation-assurance-und.md new file mode 100644 index 0000000..d7b06d1 --- /dev/null +++ b/papers/items/2026-2603-28900-robust-multi-agent-reinforcement-learning-for-small-uas-separation-assurance-und.md @@ -0,0 +1,64 @@ +# Paper: Robust Multi-Agent Reinforcement Learning for Small UAS Separation Assurance under GPS Degradation and Spoofing + +--- +type: paper +title: Robust Multi-Agent Reinforcement Learning for Small UAS Separation Assurance under GPS Degradation and Spoofing +authors: Alex Zongo, Filippos Fotiadis, Ufuk Topcu, Peng Wei +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2603.28900 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-03-30 +updated_at: 2026-03-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO + - cs.AI + - cs.LG + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, multi-agent, world-model +- arXiv categories: cs.RO, cs.AI, cs.LG, eess.SY +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2603.28900 diff --git a/papers/items/2026-2604-02022-atbench-a-diverse-and-realistic-agent-trajectory-benchmark-for-safety-evaluation.md b/papers/items/2026-2604-02022-atbench-a-diverse-and-realistic-agent-trajectory-benchmark-for-safety-evaluation.md new file mode 100644 index 0000000..cfe6183 --- /dev/null +++ b/papers/items/2026-2604-02022-atbench-a-diverse-and-realistic-agent-trajectory-benchmark-for-safety-evaluation.md @@ -0,0 +1,62 @@ +# Paper: ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis + +--- +type: paper +title: "ATBench: A Diverse and Realistic Agent Trajectory Benchmark for Safety Evaluation and Diagnosis" +authors: Yu Li, Haoyu Luo, Yuejin Xie, Yuqian Fu, Zhonghao Yang, Shuai Shao, Qihan Ren, Wanying Qu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.02022 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-02 +updated_at: 2026-05-13 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.02022 diff --git a/papers/items/2026-2604-02155-brief-is-better-non-monotonic-chain-of-thought-budget-effects-in-function-callin.md b/papers/items/2026-2604-02155-brief-is-better-non-monotonic-chain-of-thought-budget-effects-in-function-callin.md new file mode 100644 index 0000000..755df83 --- /dev/null +++ b/papers/items/2026-2604-02155-brief-is-better-non-monotonic-chain-of-thought-budget-effects-in-function-callin.md @@ -0,0 +1,61 @@ +# Paper: Brief Is Better: Non-Monotonic Chain-of-Thought Budget Effects in Function-Calling Language Agents + +--- +type: paper +title: "Brief Is Better: Non-Monotonic Chain-of-Thought Budget Effects in Function-Calling Language Agents" +authors: Xuan Qi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.02155 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-02 +updated_at: 2026-04-02 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: function-calling, language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling, language-agent +- inferred topics: agent-evaluation, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.02155 diff --git a/papers/items/2026-2604-03098-co-evolution-of-policy-and-internal-reward-for-language-agents.md b/papers/items/2026-2604-03098-co-evolution-of-policy-and-internal-reward-for-language-agents.md new file mode 100644 index 0000000..e69c3b5 --- /dev/null +++ b/papers/items/2026-2604-03098-co-evolution-of-policy-and-internal-reward-for-language-agents.md @@ -0,0 +1,63 @@ +# Paper: Co-Evolution of Policy and Internal Reward for Language Agents + +--- +type: paper +title: Co-Evolution of Policy and Internal Reward for Language Agents +authors: Xinyu Wang, Hanwei Wu, Jingwei Song, Shuyuan Zhang, Jiayi Zhang, Fanqi Kong, Tung Sum Thomas Kwok, Xiao-Wen Chang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.03098 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-03 +updated_at: 2026-04-03 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.03098 diff --git a/papers/items/2026-2604-03242-draft-task-decoupled-latent-reasoning-for-agent-safety.md b/papers/items/2026-2604-03242-draft-task-decoupled-latent-reasoning-for-agent-safety.md new file mode 100644 index 0000000..2ff0c1a --- /dev/null +++ b/papers/items/2026-2604-03242-draft-task-decoupled-latent-reasoning-for-agent-safety.md @@ -0,0 +1,62 @@ +# Paper: DRAFT: Task Decoupled Latent Reasoning for Agent Safety + +--- +type: paper +title: "DRAFT: Task Decoupled Latent Reasoning for Agent Safety" +authors: Lin Wang, Junfeng Fang, Dan Zhang, Fei Shen, Xiang Wang, Tat-Seng Chua +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.03242 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-02-11 +updated_at: 2026-02-11 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, rag, reasoning, tool-use +- arXiv categories: cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.03242 diff --git a/papers/items/2026-2604-04131-profile-then-reason-bounded-semantic-complexity-for-tool-augmented-language-agen.md b/papers/items/2026-2604-04131-profile-then-reason-bounded-semantic-complexity-for-tool-augmented-language-agen.md new file mode 100644 index 0000000..aa084df --- /dev/null +++ b/papers/items/2026-2604-04131-profile-then-reason-bounded-semantic-complexity-for-tool-augmented-language-agen.md @@ -0,0 +1,62 @@ +# Paper: Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents + +--- +type: paper +title: "Profile-Then-Reason: Bounded Semantic Complexity for Tool-Augmented Language Agents" +authors: Paulo Akira F. Enabe +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.04131 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-05 +updated_at: 2026-04-05 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.04131 diff --git a/papers/items/2026-2604-04426-shieldnet-network-level-guardrails-against-emerging-supply-chain-injections-in-a.md b/papers/items/2026-2604-04426-shieldnet-network-level-guardrails-against-emerging-supply-chain-injections-in-a.md new file mode 100644 index 0000000..f86b31e --- /dev/null +++ b/papers/items/2026-2604-04426-shieldnet-network-level-guardrails-against-emerging-supply-chain-injections-in-a.md @@ -0,0 +1,60 @@ +# Paper: ShieldNet: Network-Level Guardrails against Emerging Supply-Chain Injections in Agentic Systems + +--- +type: paper +title: "ShieldNet: Network-Level Guardrails against Emerging Supply-Chain Injections in Agentic Systems" +authors: Zhuowen Yuan, Zhaorun Chen, Zhen Xiang, Nathaniel D. Bastian, Seyyed Hadi Hashemi, Chaowei Xiao, Wenbo Guo, Bo Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.04426 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-06 +updated_at: 2026-04-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.04426 diff --git a/papers/items/2026-2604-06762-arulecon-agentic-security-rule-conversion.md b/papers/items/2026-2604-06762-arulecon-agentic-security-rule-conversion.md new file mode 100644 index 0000000..67adcc6 --- /dev/null +++ b/papers/items/2026-2604-06762-arulecon-agentic-security-rule-conversion.md @@ -0,0 +1,60 @@ +# Paper: ARuleCon: Agentic Security Rule Conversion + +--- +type: paper +title: "ARuleCon: Agentic Security Rule Conversion" +authors: Ming Xu, Hongtai Wang, Yanpei Guo, Zhengmin Yu, Weili Han, Hoon Wei Lim, Jin Song Dong, Jiaheng Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.06762 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-08 +updated_at: 2026-04-08 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, rag +- arXiv categories: cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.06762 diff --git a/papers/items/2026-2604-06972-differentiable-environment-trajectory-co-optimization-for-safe-multi-agent-navig.md b/papers/items/2026-2604-06972-differentiable-environment-trajectory-co-optimization-for-safe-multi-agent-navig.md new file mode 100644 index 0000000..51127a1 --- /dev/null +++ b/papers/items/2026-2604-06972-differentiable-environment-trajectory-co-optimization-for-safe-multi-agent-navig.md @@ -0,0 +1,65 @@ +# Paper: Differentiable Environment-Trajectory Co-Optimization for Safe Multi-Agent Navigation + +--- +type: paper +title: Differentiable Environment-Trajectory Co-Optimization for Safe Multi-Agent Navigation +authors: Zhan Gao, Gabriele Fadini, Stelian Coros, Amanda Prorok +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.06972 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-08 +updated_at: 2026-04-08 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - embodied-agent + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, embodied-agent, multi-agent, rag, tool-use +- arXiv categories: cs.RO, cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.06972 diff --git a/papers/items/2026-2604-08388-awakening-the-sleeping-agent-lean-specific-agentic-data-reactivates-general-tool.md b/papers/items/2026-2604-08388-awakening-the-sleeping-agent-lean-specific-agentic-data-reactivates-general-tool.md new file mode 100644 index 0000000..cac47ff --- /dev/null +++ b/papers/items/2026-2604-08388-awakening-the-sleeping-agent-lean-specific-agentic-data-reactivates-general-tool.md @@ -0,0 +1,59 @@ +# Paper: Awakening the Sleeping Agent: Lean-Specific Agentic Data Reactivates General Tool Use in Goedel Prover + +--- +type: paper +title: "Awakening the Sleeping Agent: Lean-Specific Agentic Data Reactivates General Tool Use in Goedel Prover" +authors: Jui-Hui Chung, Hongzhou Lin, Lai Jiang, Shange Tang, Chi Jin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.08388 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-09 +updated_at: 2026-04-09 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.08388 diff --git a/papers/items/2026-2604-10577-the-blind-spot-of-agent-safety-how-benign-user-instructions-expose-critical-vuln.md b/papers/items/2026-2604-10577-the-blind-spot-of-agent-safety-how-benign-user-instructions-expose-critical-vuln.md new file mode 100644 index 0000000..2805097 --- /dev/null +++ b/papers/items/2026-2604-10577-the-blind-spot-of-agent-safety-how-benign-user-instructions-expose-critical-vuln.md @@ -0,0 +1,63 @@ +# Paper: The Blind Spot of Agent Safety: How Benign User Instructions Expose Critical Vulnerabilities in Computer-Use Agents + +--- +type: paper +title: "The Blind Spot of Agent Safety: How Benign User Instructions Expose Critical Vulnerabilities in Computer-Use Agents" +authors: Xuwei Ding, Skylar Zhai, Linxin Song, Jiate Li, Taiwei Shi, Nicholas Meade, Siva Reddy, Jian Kang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.10577 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-12 +updated_at: 2026-04-17 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.10577 diff --git a/papers/items/2026-2604-11557-unitoolcall-unifying-tool-use-representation-data-and-evaluation-for-llm-agents.md b/papers/items/2026-2604-11557-unitoolcall-unifying-tool-use-representation-data-and-evaluation-for-llm-agents.md new file mode 100644 index 0000000..f947c26 --- /dev/null +++ b/papers/items/2026-2604-11557-unitoolcall-unifying-tool-use-representation-data-and-evaluation-for-llm-agents.md @@ -0,0 +1,60 @@ +# Paper: UniToolCall: Unifying Tool-Use Representation, Data, and Evaluation for LLM Agents + +--- +type: paper +title: "UniToolCall: Unifying Tool-Use Representation, Data, and Evaluation for LLM Agents" +authors: Yijuan Liang, Xinghao Chen, Yifan Ge, Ziyi Wu, Hao Wu, Changyu Zeng, Wei Xing, Xiaoyu Shen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.11557 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-13 +updated_at: 2026-05-25 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.11557 diff --git a/papers/items/2026-2604-12986-parallax-why-ai-agents-that-think-must-never-act.md b/papers/items/2026-2604-12986-parallax-why-ai-agents-that-think-must-never-act.md new file mode 100644 index 0000000..d8e9605 --- /dev/null +++ b/papers/items/2026-2604-12986-parallax-why-ai-agents-that-think-must-never-act.md @@ -0,0 +1,63 @@ +# Paper: Parallax: Why AI Agents That Think Must Never Act + +--- +type: paper +title: "Parallax: Why AI Agents That Think Must Never Act" +authors: Joel Fokou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.12986 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-14 +updated_at: 2026-04-14 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.12986 diff --git a/papers/items/2026-2604-13298-can-agents-secure-hardware-evaluating-agentic-llm-driven-obfuscation-for-ip-prot.md b/papers/items/2026-2604-13298-can-agents-secure-hardware-evaluating-agentic-llm-driven-obfuscation-for-ip-prot.md new file mode 100644 index 0000000..a3dc60c --- /dev/null +++ b/papers/items/2026-2604-13298-can-agents-secure-hardware-evaluating-agentic-llm-driven-obfuscation-for-ip-prot.md @@ -0,0 +1,61 @@ +# Paper: Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection + +--- +type: paper +title: Can Agents Secure Hardware? Evaluating Agentic LLM-Driven Obfuscation for IP Protection +authors: Sujan Ghimire, Parsa Mirfasihi, Muhtasim Alam Chowdhury, Veeramani Pugazhenthi, Harish Kumar Dharavath, Farshad Firouzi, Rozhin Yasaei, Pratik Satam, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.13298 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-14 +updated_at: 2026-04-14 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, planning, rag +- arXiv categories: cs.CR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.13298 diff --git a/papers/items/2026-2604-13536-don-t-let-ai-agents-yolo-your-files-shifting-information-and-control-to-filesyst.md b/papers/items/2026-2604-13536-don-t-let-ai-agents-yolo-your-files-shifting-information-and-control-to-filesyst.md new file mode 100644 index 0000000..ee4fb0c --- /dev/null +++ b/papers/items/2026-2604-13536-don-t-let-ai-agents-yolo-your-files-shifting-information-and-control-to-filesyst.md @@ -0,0 +1,61 @@ +# Paper: Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy + +--- +type: paper +title: "Don't Let AI Agents YOLO Your Files: Shifting Information and Control to Filesystems for Agent Safety and Autonomy" +authors: Shawn Wanxiang Zhong, Junxuan Liao, Jing Liu, Mai Zheng, Andrea C. Arpaci-Dusseau, Remzi H. Arpaci-Dusseau +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.13536 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-15 +updated_at: 2026-04-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.OS +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.OS +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.13536 diff --git a/papers/items/2026-2604-13954-hintbench-horizon-agent-intrinsic-non-attack-trajectory-benchmark.md b/papers/items/2026-2604-13954-hintbench-horizon-agent-intrinsic-non-attack-trajectory-benchmark.md new file mode 100644 index 0000000..1ddb7b7 --- /dev/null +++ b/papers/items/2026-2604-13954-hintbench-horizon-agent-intrinsic-non-attack-trajectory-benchmark.md @@ -0,0 +1,62 @@ +# Paper: HINTBench: Horizon-agent Intrinsic Non-attack Trajectory Benchmark + +--- +type: paper +title: "HINTBench: Horizon-agent Intrinsic Non-attack Trajectory Benchmark" +authors: Jiacheng Wang, Jinchang Hou, Fabian Wang, Ping Jian, Chenfu Bao, Zhonghou Lv +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.13954 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-15 +updated_at: 2026-04-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, planning, rag +- arXiv categories: cs.LG, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.13954 diff --git a/papers/items/2026-2604-14399-spacemind-a-modular-and-self-evolving-embodied-vision-language-agent-framework-f.md b/papers/items/2026-2604-14399-spacemind-a-modular-and-self-evolving-embodied-vision-language-agent-framework-f.md new file mode 100644 index 0000000..9005f4e --- /dev/null +++ b/papers/items/2026-2604-14399-spacemind-a-modular-and-self-evolving-embodied-vision-language-agent-framework-f.md @@ -0,0 +1,63 @@ +# Paper: SpaceMind: A Modular and Self-Evolving Embodied Vision-Language Agent Framework for Autonomous On-orbit Servicing + +--- +type: paper +title: "SpaceMind: A Modular and Self-Evolving Embodied Vision-Language Agent Framework for Autonomous On-orbit Servicing" +authors: Aodi Wu, Haodong Han, Xubo Luo, Ruisuo Wang, Shan He, Xue Wan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.14399 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-15 +updated_at: 2026-04-15 +status: queued +relevance: high +topics: + - embodied-agent + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO + - cs.AI + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: embodied-agent, reasoning, tool-use, world-model +- arXiv categories: cs.RO, cs.AI, eess.SY +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.14399 diff --git a/papers/items/2026-2604-15415-harmfulskillbench-how-do-harmful-skills-weaponize-your-agents.md b/papers/items/2026-2604-15415-harmfulskillbench-how-do-harmful-skills-weaponize-your-agents.md new file mode 100644 index 0000000..0a43d76 --- /dev/null +++ b/papers/items/2026-2604-15415-harmfulskillbench-how-do-harmful-skills-weaponize-your-agents.md @@ -0,0 +1,62 @@ +# Paper: HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents? + +--- +type: paper +title: "HarmfulSkillBench: How Do Harmful Skills Weaponize Your Agents?" +authors: Yukun Jiang, Yage Zhang, Michael Backes, Xinyue Shen, Yang Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.15415 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-16 +updated_at: 2026-04-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.15415 diff --git a/papers/items/2026-2604-15579-don-t-make-models-guess-security-and-safety-symbolic-guardrails-for-domain-speci.md b/papers/items/2026-2604-15579-don-t-make-models-guess-security-and-safety-symbolic-guardrails-for-domain-speci.md new file mode 100644 index 0000000..25e0670 --- /dev/null +++ b/papers/items/2026-2604-15579-don-t-make-models-guess-security-and-safety-symbolic-guardrails-for-domain-speci.md @@ -0,0 +1,63 @@ +# Paper: Don't Make Models Guess Security and Safety: Symbolic Guardrails for Domain-Specific AI Agents + +--- +type: paper +title: "Don't Make Models Guess Security and Safety: Symbolic Guardrails for Domain-Specific AI Agents" +authors: Yining Hong, Yining She, Eunsuk Kang, Christopher S. Timperley, Christian Kästner +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.15579 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-16 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.SE, cs.AI, cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.15579 diff --git a/papers/items/2026-2604-16706-evaluating-tool-using-language-agents-judge-reliability-propagation-cascades-and.md b/papers/items/2026-2604-16706-evaluating-tool-using-language-agents-judge-reliability-propagation-cascades-and.md new file mode 100644 index 0000000..ae4e51e --- /dev/null +++ b/papers/items/2026-2604-16706-evaluating-tool-using-language-agents-judge-reliability-propagation-cascades-and.md @@ -0,0 +1,61 @@ +# Paper: Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench + +--- +type: paper +title: "Evaluating Tool-Using Language Agents: Judge Reliability, Propagation Cascades, and Runtime Mitigation in AgentProp-Bench" +authors: Bhaskar Gurram +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.16706 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-17 +updated_at: 2026-04-17 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.AI, cs.CL, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.16706 diff --git a/papers/items/2026-2604-17562-safeagent-a-runtime-protection-architecture-for-agentic-systems.md b/papers/items/2026-2604-17562-safeagent-a-runtime-protection-architecture-for-agentic-systems.md new file mode 100644 index 0000000..2340230 --- /dev/null +++ b/papers/items/2026-2604-17562-safeagent-a-runtime-protection-architecture-for-agentic-systems.md @@ -0,0 +1,65 @@ +# Paper: SafeAgent: A Runtime Protection Architecture for Agentic Systems + +--- +type: paper +title: "SafeAgent: A Runtime Protection Architecture for Agentic Systems" +authors: Hailin Liu, Eugene Ilyushin, Jie Ni, Min Zhu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.17562 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-19 +updated_at: 2026-04-19 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.MA +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.17562 diff --git a/papers/items/2026-2604-18658-owner-harm-a-missing-threat-model-for-ai-agent-safety.md b/papers/items/2026-2604-18658-owner-harm-a-missing-threat-model-for-ai-agent-safety.md new file mode 100644 index 0000000..039d50d --- /dev/null +++ b/papers/items/2026-2604-18658-owner-harm-a-missing-threat-model-for-ai-agent-safety.md @@ -0,0 +1,64 @@ +# Paper: Owner-Harm: A Missing Threat Model for AI Agent Safety + +--- +type: paper +title: "Owner-Harm: A Missing Threat Model for AI Agent Safety" +authors: Dongcheng Zhang, Yiqing Jiang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.18658 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-20 +updated_at: 2026-04-20 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, rag, reasoning, tool-use +- arXiv categories: cs.CR, cs.AI, cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.18658 diff --git a/papers/items/2026-2604-18718-towards-optimal-agentic-architectures-for-offensive-security-tasks.md b/papers/items/2026-2604-18718-towards-optimal-agentic-architectures-for-offensive-security-tasks.md new file mode 100644 index 0000000..00f869b --- /dev/null +++ b/papers/items/2026-2604-18718-towards-optimal-agentic-architectures-for-offensive-security-tasks.md @@ -0,0 +1,62 @@ +# Paper: Towards Optimal Agentic Architectures for Offensive Security Tasks + +--- +type: paper +title: Towards Optimal Agentic Architectures for Offensive Security Tasks +authors: Isaac David, Arthur Gervais +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.18718 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-20 +updated_at: 2026-04-20 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.18718 diff --git a/papers/items/2026-2604-18847-human-guided-harm-recovery-for-computer-use-agents.md b/papers/items/2026-2604-18847-human-guided-harm-recovery-for-computer-use-agents.md new file mode 100644 index 0000000..6303a84 --- /dev/null +++ b/papers/items/2026-2604-18847-human-guided-harm-recovery-for-computer-use-agents.md @@ -0,0 +1,65 @@ +# Paper: Human-Guided Harm Recovery for Computer Use Agents + +--- +type: paper +title: Human-Guided Harm Recovery for Computer Use Agents +authors: Christy Li, Sky CH-Wang, Andi Peng, Andreea Bobu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.18847 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-20 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, planning, rag, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.18847 diff --git a/papers/items/2026-2604-19821-jtpro-a-joint-tool-prompt-reflective-optimization-framework-for-language-agents.md b/papers/items/2026-2604-19821-jtpro-a-joint-tool-prompt-reflective-optimization-framework-for-language-agents.md new file mode 100644 index 0000000..72651bd --- /dev/null +++ b/papers/items/2026-2604-19821-jtpro-a-joint-tool-prompt-reflective-optimization-framework-for-language-agents.md @@ -0,0 +1,62 @@ +# Paper: JTPRO: A Joint Tool-Prompt Reflective Optimization Framework for Language Agents + +--- +type: paper +title: "JTPRO: A Joint Tool-Prompt Reflective Optimization Framework for Language Agents" +authors: Sandip Ghoshal, Anshul Mittal, Jyotika Singh, Miguel Ballesteros, Weiyi Sun, Fang Tu, Shailender Singh, Yassine Benajiba, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.19821 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-20 +updated_at: 2026-04-20 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, computer-use, reasoning, tool-use +- arXiv categories: cs.AI, cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.19821 diff --git a/papers/items/2026-2604-19844-if-you-re-waiting-for-a-sign-that-might-not-be-it-mitigating-trust-boundary-conf.md b/papers/items/2026-2604-19844-if-you-re-waiting-for-a-sign-that-might-not-be-it-mitigating-trust-boundary-conf.md new file mode 100644 index 0000000..aed1dd6 --- /dev/null +++ b/papers/items/2026-2604-19844-if-you-re-waiting-for-a-sign-that-might-not-be-it-mitigating-trust-boundary-conf.md @@ -0,0 +1,62 @@ +# Paper: If you're waiting for a sign... that might not be it! Mitigating Trust Boundary Confusion from Visual Injections on Vision-Language Agentic Systems + +--- +type: paper +title: "If you're waiting for a sign... that might not be it! Mitigating Trust Boundary Confusion from Visual Injections on Vision-Language Agentic Systems" +authors: Jiamin Chang, Minhui Xue, Ruoxi Sun, Shuchao Pang, Salil S. Kanhere, Hammond Pearce +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.19844 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-21 +updated_at: 2026-04-21 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - embodied-agent + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, agent-safety, embodied-agent, multi-agent +- arXiv categories: cs.CV, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.19844 diff --git a/papers/items/2026-2604-20994-breaking-mcp-with-function-hijacking-attacks-novel-threats-for-function-calling-.md b/papers/items/2026-2604-20994-breaking-mcp-with-function-hijacking-attacks-novel-threats-for-function-calling-.md new file mode 100644 index 0000000..65ee24c --- /dev/null +++ b/papers/items/2026-2604-20994-breaking-mcp-with-function-hijacking-attacks-novel-threats-for-function-calling-.md @@ -0,0 +1,62 @@ +# Paper: Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models + +--- +type: paper +title: "Breaking MCP with Function Hijacking Attacks: Novel Threats for Function Calling and Agentic Models" +authors: Yannis Belkhiter, Giulio Zizzo, Sergio Maffeis, Seshu Tirupathi, John D. Kelleher +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.20994 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-22 +updated_at: 2026-04-22 +status: queued +relevance: high +topics: + - agent-safety + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-safety, reasoning, tool-use +- arXiv categories: cs.CR, cs.AI, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.20994 diff --git a/papers/items/2026-2604-21190-spatio-adaptive-test-time-orchestration-of-vision-language-agents-for-spatial-re.md b/papers/items/2026-2604-21190-spatio-adaptive-test-time-orchestration-of-vision-language-agents-for-spatial-re.md new file mode 100644 index 0000000..db04c41 --- /dev/null +++ b/papers/items/2026-2604-21190-spatio-adaptive-test-time-orchestration-of-vision-language-agents-for-spatial-re.md @@ -0,0 +1,61 @@ +# Paper: SpatiO: Adaptive Test-Time Orchestration of Vision-Language Agents for Spatial Reasoning + +--- +type: paper +title: "SpatiO: Adaptive Test-Time Orchestration of Vision-Language Agents for Spatial Reasoning" +authors: Chan Yeong Hwang, Miso Choi, Sunghyun On, Jinkyu Kim, Jungbeom Lee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.21190 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-23 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, multi-agent, rag, reasoning +- arXiv categories: cs.CV +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.21190 diff --git a/papers/items/2026-2604-22879-beyond-single-agent-alignment-preventing-context-fragmented-violations-in-multi-.md b/papers/items/2026-2604-22879-beyond-single-agent-alignment-preventing-context-fragmented-violations-in-multi-.md new file mode 100644 index 0000000..b379021 --- /dev/null +++ b/papers/items/2026-2604-22879-beyond-single-agent-alignment-preventing-context-fragmented-violations-in-multi-.md @@ -0,0 +1,67 @@ +# Paper: Beyond Single-Agent Alignment: Preventing Context-Fragmented Violations in Multi-Agent Systems + +--- +type: paper +title: "Beyond Single-Agent Alignment: Preventing Context-Fragmented Violations in Multi-Agent Systems" +authors: Jie Wu, Ming Gong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.22879 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-24 +updated_at: 2026-04-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.AI + - cs.CR + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, tool-use, workflow-agent, world-model +- arXiv categories: cs.MA, cs.AI, cs.CR, cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.22879 diff --git a/papers/items/2026-2604-23210-discovering-agentic-safety-specifications-from-1-bit-danger-signals.md b/papers/items/2026-2604-23210-discovering-agentic-safety-specifications-from-1-bit-danger-signals.md new file mode 100644 index 0000000..2dbae72 --- /dev/null +++ b/papers/items/2026-2604-23210-discovering-agentic-safety-specifications-from-1-bit-danger-signals.md @@ -0,0 +1,64 @@ +# Paper: Discovering Agentic Safety Specifications from 1-Bit Danger Signals + +--- +type: paper +title: Discovering Agentic Safety Specifications from 1-Bit Danger Signals +authors: Víctor Gallego +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.23210 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-25 +updated_at: 2026-04-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.23210 diff --git a/papers/items/2026-2604-23374-ghost-in-the-agent-redefining-information-flow-tracking-for-llm-agents.md b/papers/items/2026-2604-23374-ghost-in-the-agent-redefining-information-flow-tracking-for-llm-agents.md new file mode 100644 index 0000000..47f1fb1 --- /dev/null +++ b/papers/items/2026-2604-23374-ghost-in-the-agent-redefining-information-flow-tracking-for-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents + +--- +type: paper +title: "Ghost in the Agent: Redefining Information Flow Tracking for LLM Agents" +authors: Yuandao Cai, Wensheng Tang, Cheng Wen, Shengchao Qin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.23374 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-25 +updated_at: 2026-04-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, memory, reasoning, tool-use +- arXiv categories: cs.CR +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.23374 diff --git a/papers/items/2026-2604-23459-architecture-matters-for-multi-agent-security.md b/papers/items/2026-2604-23459-architecture-matters-for-multi-agent-security.md new file mode 100644 index 0000000..b127baf --- /dev/null +++ b/papers/items/2026-2604-23459-architecture-matters-for-multi-agent-security.md @@ -0,0 +1,65 @@ +# Paper: Architecture Matters for Multi-Agent Security + +--- +type: paper +title: Architecture Matters for Multi-Agent Security +authors: Ben Hagag, William L. Anderson, Christian Schroeder de Witt, Sarah Scheffler +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.23459 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-25 +updated_at: 2026-04-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - multi-agent + - planning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.CR + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, multi-agent, planning +- arXiv categories: cs.MA, cs.CR, cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.23459 diff --git a/papers/items/2026-2604-24212-empowering-autonomous-debugging-agents-with-efficient-dynamic-analysis.md b/papers/items/2026-2604-24212-empowering-autonomous-debugging-agents-with-efficient-dynamic-analysis.md new file mode 100644 index 0000000..169b7fb --- /dev/null +++ b/papers/items/2026-2604-24212-empowering-autonomous-debugging-agents-with-efficient-dynamic-analysis.md @@ -0,0 +1,63 @@ +# Paper: Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis + +--- +type: paper +title: Empowering Autonomous Debugging Agents with Efficient Dynamic Analysis +authors: Jiahong Xiang, Xiaoyang Xu, Xiaopan Chu, Hongliang Tian, Yuqun Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.24212 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-27 +updated_at: 2026-04-27 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - embodied-agent + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, coding-agent, embodied-agent, memory, rag, tool-use +- arXiv categories: cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.24212 diff --git a/papers/items/2026-2604-24826-a-comparative-evaluation-of-ai-agent-security-guardrails.md b/papers/items/2026-2604-24826-a-comparative-evaluation-of-ai-agent-security-guardrails.md new file mode 100644 index 0000000..6259e60 --- /dev/null +++ b/papers/items/2026-2604-24826-a-comparative-evaluation-of-ai-agent-security-guardrails.md @@ -0,0 +1,61 @@ +# Paper: A Comparative Evaluation of AI Agent Security Guardrails + +--- +type: paper +title: A Comparative Evaluation of AI Agent Security Guardrails +authors: Qi Li, Jiu Li, Pingtao Wei, Jianjun Xu, Xueyi Wei, Jiwei Shi, Xuan Zhang, Yanhui Yang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.24826 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-27 +updated_at: 2026-04-27 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.24826 diff --git a/papers/items/2026-2604-25135-fama-failure-aware-meta-agentic-framework-for-open-source-llms-in-interactive-to.md b/papers/items/2026-2604-25135-fama-failure-aware-meta-agentic-framework-for-open-source-llms-in-interactive-to.md new file mode 100644 index 0000000..3f233dc --- /dev/null +++ b/papers/items/2026-2604-25135-fama-failure-aware-meta-agentic-framework-for-open-source-llms-in-interactive-to.md @@ -0,0 +1,59 @@ +# Paper: FAMA: Failure-Aware Meta-Agentic Framework for Open-Source LLMs in Interactive Tool Use Environments + +--- +type: paper +title: "FAMA: Failure-Aware Meta-Agentic Framework for Open-Source LLMs in Interactive Tool Use Environments" +authors: Amir Saeidi, Venkatesh Mishra, Souradeep Mukhopadhyay, Gaowen Liu, Ali Payani, Jayanth Srinivasa, Chitta Baral +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.25135 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-28 +updated_at: 2026-04-28 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.25135 diff --git a/papers/items/2026-2604-25318-cutscene-agent-an-llm-agent-framework-for-automated-3d-cutscene-generation.md b/papers/items/2026-2604-25318-cutscene-agent-an-llm-agent-framework-for-automated-3d-cutscene-generation.md new file mode 100644 index 0000000..8465e61 --- /dev/null +++ b/papers/items/2026-2604-25318-cutscene-agent-an-llm-agent-framework-for-automated-3d-cutscene-generation.md @@ -0,0 +1,64 @@ +# Paper: Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation + +--- +type: paper +title: "Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation" +authors: Lanshan He, Haozhou Pang, Qi Gan, Xin Shen, Ziwei Zhang, Yibo Liu, Gang Fang, Bo Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.25318 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-28 +updated_at: 2026-04-28 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.GR + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.GR, cs.AI, cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.25318 diff --git a/papers/items/2026-2604-25555-from-crud-to-autonomous-agents-formal-validation-and-zero-trust-security-for-sem.md b/papers/items/2026-2604-25555-from-crud-to-autonomous-agents-formal-validation-and-zero-trust-security-for-sem.md new file mode 100644 index 0000000..85931ab --- /dev/null +++ b/papers/items/2026-2604-25555-from-crud-to-autonomous-agents-formal-validation-and-zero-trust-security-for-sem.md @@ -0,0 +1,63 @@ +# Paper: From CRUD to Autonomous Agents: Formal Validation and Zero-Trust Security for Semantic Gateways in AI-Native Enterprise Systems + +--- +type: paper +title: "From CRUD to Autonomous Agents: Formal Validation and Zero-Trust Security for Semantic Gateways in AI-Native Enterprise Systems" +authors: Ignacio Peyrano +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.25555 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-28 +updated_at: 2026-04-28 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.25555 diff --git a/papers/items/2026-2604-26274-enforcing-benign-trajectories-a-behavioral-firewall-for-structured-workflow-ai-a.md b/papers/items/2026-2604-26274-enforcing-benign-trajectories-a-behavioral-firewall-for-structured-workflow-ai-a.md new file mode 100644 index 0000000..6524538 --- /dev/null +++ b/papers/items/2026-2604-26274-enforcing-benign-trajectories-a-behavioral-firewall-for-structured-workflow-ai-a.md @@ -0,0 +1,63 @@ +# Paper: Enforcing Benign Trajectories: A Behavioral Firewall for Structured-Workflow AI Agents + +--- +type: paper +title: "Enforcing Benign Trajectories: A Behavioral Firewall for Structured-Workflow AI Agents" +authors: Hung Dang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.26274 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-29 +updated_at: 2026-04-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, rag, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.26274 diff --git a/papers/items/2026-2604-26959-careguardai-context-aware-multi-agent-guardrails-for-clinical-safety-hallucinati.md b/papers/items/2026-2604-26959-careguardai-context-aware-multi-agent-guardrails-for-clinical-safety-hallucinati.md new file mode 100644 index 0000000..d1b5966 --- /dev/null +++ b/papers/items/2026-2604-26959-careguardai-context-aware-multi-agent-guardrails-for-clinical-safety-hallucinati.md @@ -0,0 +1,63 @@ +# Paper: CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs + +--- +type: paper +title: "CareGuardAI: Context-Aware Multi-Agent Guardrails for Clinical Safety & Hallucination Mitigation in Patient-Facing LLMs" +authors: Elham Nasarian, Abhilash Neog, Kwok-Leung Tsui, Niyousha HosseiniChimeh +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.26959 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-07 +updated_at: 2026-04-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, multi-agent, tool-use +- arXiv categories: cs.CY, cs.AI, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.26959 diff --git a/papers/items/2026-2604-27092-end-to-end-autonomous-scientific-discovery-on-a-real-optical-platform.md b/papers/items/2026-2604-27092-end-to-end-autonomous-scientific-discovery-on-a-real-optical-platform.md new file mode 100644 index 0000000..59961e9 --- /dev/null +++ b/papers/items/2026-2604-27092-end-to-end-autonomous-scientific-discovery-on-a-real-optical-platform.md @@ -0,0 +1,63 @@ +# Paper: End-to-end autonomous scientific discovery on a real optical platform + +--- +type: paper +title: End-to-end autonomous scientific discovery on a real optical platform +authors: Shuxing Yang, Fujia Chen, Rui Zhao, Junyao Wu, Yize Wang, Haiyao Luo, Ning Han, Qiaolu Chen, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.27092 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-29 +updated_at: 2026-04-29 +status: queued +relevance: high +topics: + - memory + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - physics.optics +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: memory, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, physics.optics +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.27092 diff --git a/papers/items/2026-2604-27464-security-attack-and-defense-strategies-for-autonomous-agent-frameworks-a-layered.md b/papers/items/2026-2604-27464-security-attack-and-defense-strategies-for-autonomous-agent-frameworks-a-layered.md new file mode 100644 index 0000000..18fbc31 --- /dev/null +++ b/papers/items/2026-2604-27464-security-attack-and-defense-strategies-for-autonomous-agent-frameworks-a-layered.md @@ -0,0 +1,63 @@ +# Paper: Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study + +--- +type: paper +title: "Security Attack and Defense Strategies for Autonomous Agent Frameworks: A Layered Review with OpenClaw as a Case Study" +authors: Luyao Xu, Xiang Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.27464 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-30 +updated_at: 2026-04-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety, autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, planning, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.27464 diff --git a/papers/items/2026-2604-27699-bridging-values-and-behavior-a-hierarchical-framework-for-proactive-embodied-age.md b/papers/items/2026-2604-27699-bridging-values-and-behavior-a-hierarchical-framework-for-proactive-embodied-age.md new file mode 100644 index 0000000..20debb5 --- /dev/null +++ b/papers/items/2026-2604-27699-bridging-values-and-behavior-a-hierarchical-framework-for-proactive-embodied-age.md @@ -0,0 +1,64 @@ +# Paper: Bridging Values and Behavior: A Hierarchical Framework for Proactive Embodied Agents + +--- +type: paper +title: "Bridging Values and Behavior: A Hierarchical Framework for Proactive Embodied Agents" +authors: Chunhui Zhang, Yuxuan Wang, Aoyang Qin, Yi-Long Lu, Kunlun Wu, Yizhou Wang, Wei Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.27699 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-30 +updated_at: 2026-04-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - embodied-agent + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, embodied-agent, memory, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.27699 diff --git a/papers/items/2026-2604-27859-rethinking-agentic-reinforcement-learning-in-large-language-models.md b/papers/items/2026-2604-27859-rethinking-agentic-reinforcement-learning-in-large-language-models.md new file mode 100644 index 0000000..49f09b6 --- /dev/null +++ b/papers/items/2026-2604-27859-rethinking-agentic-reinforcement-learning-in-large-language-models.md @@ -0,0 +1,62 @@ +# Paper: Rethinking Agentic Reinforcement Learning In Large Language Models + +--- +type: paper +title: Rethinking Agentic Reinforcement Learning In Large Language Models +authors: Fangming Cui, Ruixiao Zhu, Cheng Fang, Sunan Li, Jiahong Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.27859 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-30 +updated_at: 2026-05-15 +status: queued +relevance: high +topics: + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.ET +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: memory, planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.ET +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.27859 diff --git a/papers/items/2026-2604-28157-flashrt-towards-computationally-and-memory-efficient-red-teaming-for-prompt-inje.md b/papers/items/2026-2604-28157-flashrt-towards-computationally-and-memory-efficient-red-teaming-for-prompt-inje.md new file mode 100644 index 0000000..bb39447 --- /dev/null +++ b/papers/items/2026-2604-28157-flashrt-towards-computationally-and-memory-efficient-red-teaming-for-prompt-inje.md @@ -0,0 +1,62 @@ +# Paper: FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption + +--- +type: paper +title: "FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption" +authors: Yanting Wang, Chenlong Yin, Ying Chen, Jinyuan Jia +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2604.28157 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-30 +updated_at: 2026-04-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use +- arXiv categories: cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2604.28157 diff --git a/papers/items/2026-2605-00081-alignment-contracts-for-agentic-security-systems.md b/papers/items/2026-2605-00081-alignment-contracts-for-agentic-security-systems.md new file mode 100644 index 0000000..5c5269f --- /dev/null +++ b/papers/items/2026-2605-00081-alignment-contracts-for-agentic-security-systems.md @@ -0,0 +1,63 @@ +# Paper: Alignment Contracts for Agentic Security Systems + +--- +type: paper +title: Alignment Contracts for Agentic Security Systems +authors: Isaac David, Marco Guarnieri, Arthur Gervais +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.00081 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-30 +updated_at: 2026-04-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.LO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, planning, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.LO +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.00081 diff --git a/papers/items/2026-2605-00741-self-adaptive-multi-agent-llm-based-security-pattern-selection-for-iot-systems.md b/papers/items/2026-2605-00741-self-adaptive-multi-agent-llm-based-security-pattern-selection-for-iot-systems.md new file mode 100644 index 0000000..bc500cc --- /dev/null +++ b/papers/items/2026-2605-00741-self-adaptive-multi-agent-llm-based-security-pattern-selection-for-iot-systems.md @@ -0,0 +1,62 @@ +# Paper: Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems + +--- +type: paper +title: Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems +authors: Saeid Jamshidi, Foutse Khomh, Carol Fung, Kawser Wazed Nafi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.00741 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-01 +updated_at: 2026-05-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, multi-agent, reasoning, tool-use +- arXiv categories: cs.CR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.00741 diff --git a/papers/items/2026-2605-00845-graph-query-generation-with-constraint-guided-large-language-agents.md b/papers/items/2026-2605-00845-graph-query-generation-with-constraint-guided-large-language-agents.md new file mode 100644 index 0000000..a5e1758 --- /dev/null +++ b/papers/items/2026-2605-00845-graph-query-generation-with-constraint-guided-large-language-agents.md @@ -0,0 +1,63 @@ +# Paper: Graph Query Generation with Constraint-guided Large Language Agents + +--- +type: paper +title: Graph Query Generation with Constraint-guided Large Language Agents +authors: Mengying Wang, Nicolaas Jedema, Rahul Pandey, RaviKiran Krishnan, Jens Lehmann, Yinghui Wu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.00845 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-09 +updated_at: 2026-04-09 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DB + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, computer-use, reasoning, workflow-agent +- arXiv categories: cs.DB, cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.00845 diff --git a/papers/items/2026-2605-01101-virtual-speech-therapist-a-clinician-in-the-loop-ai-speech-therapy-agent-for-per.md b/papers/items/2026-2605-01101-virtual-speech-therapist-a-clinician-in-the-loop-ai-speech-therapy-agent-for-per.md new file mode 100644 index 0000000..78129f3 --- /dev/null +++ b/papers/items/2026-2605-01101-virtual-speech-therapist-a-clinician-in-the-loop-ai-speech-therapy-agent-for-per.md @@ -0,0 +1,68 @@ +# Paper: Virtual Speech Therapist: A Clinician-in-the-Loop AI Speech Therapy Agent for Personalized and Supervised Therapy + +--- +type: paper +title: "Virtual Speech Therapist: A Clinician-in-the-Loop AI Speech Therapy Agent for Personalized and Supervised Therapy" +authors: Shakeel Sheikh, Patrick Marmaroli, MD Sahidullah, Slim Ouni, Fabrice Hirsch, Goncalo Leal, Bjorn W Schuller +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.01101 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-01 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.SD + - eess.AS +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CL, cs.SD, eess.AS +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.01101 diff --git a/papers/items/2026-2605-01644-toward-a-principled-framework-for-agent-safety-measurement.md b/papers/items/2026-2605-01644-toward-a-principled-framework-for-agent-safety-measurement.md new file mode 100644 index 0000000..abddee1 --- /dev/null +++ b/papers/items/2026-2605-01644-toward-a-principled-framework-for-agent-safety-measurement.md @@ -0,0 +1,61 @@ +# Paper: Toward a Principled Framework for Agent Safety Measurement + +--- +type: paper +title: Toward a Principled Framework for Agent Safety Measurement +authors: Shuyi Lin, Anshuman Suri, Alina Oprea, Cheng Tan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.01644 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-02 +updated_at: 2026-05-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.CR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.01644 diff --git a/papers/items/2026-2605-02240-physicianbench-evaluating-llm-agents-in-real-world-ehr-environments.md b/papers/items/2026-2605-02240-physicianbench-evaluating-llm-agents-in-real-world-ehr-environments.md new file mode 100644 index 0000000..8c3d085 --- /dev/null +++ b/papers/items/2026-2605-02240-physicianbench-evaluating-llm-agents-in-real-world-ehr-environments.md @@ -0,0 +1,63 @@ +# Paper: PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments + +--- +type: paper +title: "PhysicianBench: Evaluating LLM Agents in Real-World EHR Environments" +authors: Ruoqi Liu, Imran Q. Mohiuddin, Austin J. Schoeffler, Kavita Renduchintala, Ashwin Nayak, Prasantha L. Vemu, Shivam C. Vedak, Kameron C. Black, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.02240 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-04 +updated_at: 2026-05-04 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.02240 diff --git a/papers/items/2026-2605-03242-enhancing-agent-safety-judgment-controlled-benchmark-rewriting-and-analogical-re.md b/papers/items/2026-2605-03242-enhancing-agent-safety-judgment-controlled-benchmark-rewriting-and-analogical-re.md new file mode 100644 index 0000000..036c251 --- /dev/null +++ b/papers/items/2026-2605-03242-enhancing-agent-safety-judgment-controlled-benchmark-rewriting-and-analogical-re.md @@ -0,0 +1,64 @@ +# Paper: Enhancing Agent Safety Judgment: Controlled Benchmark Rewriting and Analogical Reasoning for Deceptive Out-of-Distribution Scenarios + +--- +type: paper +title: "Enhancing Agent Safety Judgment: Controlled Benchmark Rewriting and Analogical Reasoning for Deceptive Out-of-Distribution Scenarios" +authors: Zuoyu Zhang, Yancheng Zhu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.03242 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-05 +updated_at: 2026-05-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.03242 diff --git a/papers/items/2026-2605-03312-memflow-intent-driven-memory-orchestration-for-small-language-model-agents.md b/papers/items/2026-2605-03312-memflow-intent-driven-memory-orchestration-for-small-language-model-agents.md new file mode 100644 index 0000000..98f0689 --- /dev/null +++ b/papers/items/2026-2605-03312-memflow-intent-driven-memory-orchestration-for-small-language-model-agents.md @@ -0,0 +1,63 @@ +# Paper: MemFlow: Intent-Driven Memory Orchestration for Small Language Model Agents + +--- +type: paper +title: "MemFlow: Intent-Driven Memory Orchestration for Small Language Model Agents" +authors: Jiayi Chen, Yingcong Li, Guiling Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.03312 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-05 +updated_at: 2026-05-05 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.MA +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.03312 diff --git a/papers/items/2026-2605-03328-llm-adam-a-generalizable-llm-agent-framework-for-pre-print-anomaly-detection-in-.md b/papers/items/2026-2605-03328-llm-adam-a-generalizable-llm-agent-framework-for-pre-print-anomaly-detection-in-.md new file mode 100644 index 0000000..a6ba0ff --- /dev/null +++ b/papers/items/2026-2605-03328-llm-adam-a-generalizable-llm-agent-framework-for-pre-print-anomaly-detection-in-.md @@ -0,0 +1,61 @@ +# Paper: LLM-ADAM: A Generalizable LLM Agent Framework for Pre-Print Anomaly Detection in Additive Manufacturing + +--- +type: paper +title: "LLM-ADAM: A Generalizable LLM Agent Framework for Pre-Print Anomaly Detection in Additive Manufacturing" +authors: Ahmadreza Eslaminia, Chuhan Cai, Cameron Smith, Ruo-Syuan Mei, Shichen Li, Rajiv Malhotra, Klara Nahrstedt, Chenhui Shao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.03328 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-05 +updated_at: 2026-05-05 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, planning +- arXiv categories: cs.LG, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.03328 diff --git a/papers/items/2026-2605-03505-lats-rca-language-agent-tree-search-for-root-cause-analysis-in-microservices.md b/papers/items/2026-2605-03505-lats-rca-language-agent-tree-search-for-root-cause-analysis-in-microservices.md new file mode 100644 index 0000000..d9f2bb4 --- /dev/null +++ b/papers/items/2026-2605-03505-lats-rca-language-agent-tree-search-for-root-cause-analysis-in-microservices.md @@ -0,0 +1,61 @@ +# Paper: LATS-RCA: Language Agent Tree Search for Root Cause Analysis in Microservices + +--- +type: paper +title: "LATS-RCA: Language Agent Tree Search for Root Cause Analysis in Microservices" +authors: Alexander Naakka, Yuqing Wang, Mika V Mäntylä +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.03505 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-05 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, computer-use, rag, reasoning +- arXiv categories: cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.03505 diff --git a/papers/items/2026-2605-04107-tscg-deterministic-tool-schema-compilation-for-agentic-llm-deployments.md b/papers/items/2026-2605-04107-tscg-deterministic-tool-schema-compilation-for-agentic-llm-deployments.md new file mode 100644 index 0000000..875e38b --- /dev/null +++ b/papers/items/2026-2605-04107-tscg-deterministic-tool-schema-compilation-for-agentic-llm-deployments.md @@ -0,0 +1,62 @@ +# Paper: TSCG: Deterministic Tool-Schema Compilation for Agentic LLM Deployments + +--- +type: paper +title: "TSCG: Deterministic Tool-Schema Compilation for Agentic LLM Deployments" +authors: Furkan Sakizli +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.04107 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-04 +updated_at: 2026-05-04 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, tool-use +- arXiv categories: cs.SE, cs.AI, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.04107 diff --git a/papers/items/2026-2605-04808-decodingtrust-agent-platform-dtap-a-controllable-and-interactive-red-teaming-pla.md b/papers/items/2026-2605-04808-decodingtrust-agent-platform-dtap-a-controllable-and-interactive-red-teaming-pla.md new file mode 100644 index 0000000..a2f1d7b --- /dev/null +++ b/papers/items/2026-2605-04808-decodingtrust-agent-platform-dtap-a-controllable-and-interactive-red-teaming-pla.md @@ -0,0 +1,64 @@ +# Paper: DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents + +--- +type: paper +title: "DecodingTrust-Agent Platform (DTap): A Controllable and Interactive Red-Teaming Platform for AI Agents" +authors: Zhaorun Chen, Xun Liu, Haibo Tong, Chengquan Guo, Yuzhou Nie, Jiawei Zhang, Mintong Kang, Chejian Xu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.04808 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-06 +updated_at: 2026-05-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - planning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, coding-agent, planning, tool-use, workflow-agent, world-model +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.04808 diff --git a/papers/items/2026-2605-05242-beyond-semantic-similarity-rethinking-retrieval-for-agentic-search-via-direct-co.md b/papers/items/2026-2605-05242-beyond-semantic-similarity-rethinking-retrieval-for-agentic-search-via-direct-co.md new file mode 100644 index 0000000..74b26d7 --- /dev/null +++ b/papers/items/2026-2605-05242-beyond-semantic-similarity-rethinking-retrieval-for-agentic-search-via-direct-co.md @@ -0,0 +1,63 @@ +# Paper: Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction + +--- +type: paper +title: "Beyond Semantic Similarity: Rethinking Retrieval for Agentic Search via Direct Corpus Interaction" +authors: Zhuofeng Li, Haoxiang Zhang, Cong Wei, Pan Lu, Ping Nie, Yi Lu, Yuyang Bai, Shangbin Feng, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.05242 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-03 +updated_at: 2026-05-03 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use +- arXiv categories: cs.IR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.05242 diff --git a/papers/items/2026-2605-05704-safeharbor-hierarchical-memory-augmented-guardrail-for-llm-agent-safety.md b/papers/items/2026-2605-05704-safeharbor-hierarchical-memory-augmented-guardrail-for-llm-agent-safety.md new file mode 100644 index 0000000..80a955b --- /dev/null +++ b/papers/items/2026-2605-05704-safeharbor-hierarchical-memory-augmented-guardrail-for-llm-agent-safety.md @@ -0,0 +1,63 @@ +# Paper: SafeHarbor: Hierarchical Memory-Augmented Guardrail for LLM Agent Safety + +--- +type: paper +title: "SafeHarbor: Hierarchical Memory-Augmented Guardrail for LLM Agent Safety" +authors: Zhe Liu, Zonghao Ying, Wenxin Zhang, Quanchen Zou, Deyue Zhang, Dongdong Yang, Xiangzheng Zhang, Hao Peng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.05704 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - memory + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: agent-safety, autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, autonomous-agent-llm +- inferred topics: agent-safety, computer-use, memory, reasoning, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.05704 diff --git a/papers/items/2026-2605-05716-more-is-not-always-better-cross-component-interference-in-llm-agent-scaffolding.md b/papers/items/2026-2605-05716-more-is-not-always-better-cross-component-interference-in-llm-agent-scaffolding.md new file mode 100644 index 0000000..a7127f9 --- /dev/null +++ b/papers/items/2026-2605-05716-more-is-not-always-better-cross-component-interference-in-llm-agent-scaffolding.md @@ -0,0 +1,64 @@ +# Paper: More Is Not Always Better: Cross-Component Interference in LLM Agent Scaffolding + +--- +type: paper +title: "More Is Not Always Better: Cross-Component Interference in LLM Agent Scaffolding" +authors: Ming Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.05716 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-05-07 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.05716 diff --git a/papers/items/2026-2605-06078-milestone-guided-policy-learning-for-long-horizon-language-agents.md b/papers/items/2026-2605-06078-milestone-guided-policy-learning-for-long-horizon-language-agents.md new file mode 100644 index 0000000..69577bb --- /dev/null +++ b/papers/items/2026-2605-06078-milestone-guided-policy-learning-for-long-horizon-language-agents.md @@ -0,0 +1,63 @@ +# Paper: Milestone-Guided Policy Learning for Long-Horizon Language Agents + +--- +type: paper +title: Milestone-Guided Policy Learning for Long-Horizon Language Agents +authors: Zixuan Wang, Yuchen Yan, Hongxing Li, Teng Pan, Dingming Li, Ruiqing Zhang, Weiming Lu, Jun Xiao, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.06078 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-05-07 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, computer-use, planning, rag, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.06078 diff --git a/papers/items/2026-2605-06713-agentic-ai-and-the-industrialization-of-cyber-offense-forecast-consequences-and-.md b/papers/items/2026-2605-06713-agentic-ai-and-the-industrialization-of-cyber-offense-forecast-consequences-and-.md new file mode 100644 index 0000000..474c789 --- /dev/null +++ b/papers/items/2026-2605-06713-agentic-ai-and-the-industrialization-of-cyber-offense-forecast-consequences-and-.md @@ -0,0 +1,64 @@ +# Paper: Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand + +--- +type: paper +title: "Agentic AI and the Industrialization of Cyber Offense: Forecast, Consequences, and Defensive Priorities for Enterprises and the Mittelstand" +authors: Christopher Koch +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.06713 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-06 +updated_at: 2026-05-06 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-safety, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, planning-agent +- inferred topics: agent-safety, computer-use, planning, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI, cs.HC +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.06713 diff --git a/papers/items/2026-2605-06716-from-storage-to-experience-a-survey-on-the-evolution-of-llm-agent-memory-mechani.md b/papers/items/2026-2605-06716-from-storage-to-experience-a-survey-on-the-evolution-of-llm-agent-memory-mechani.md new file mode 100644 index 0000000..11700a1 --- /dev/null +++ b/papers/items/2026-2605-06716-from-storage-to-experience-a-survey-on-the-evolution-of-llm-agent-memory-mechani.md @@ -0,0 +1,63 @@ +# Paper: From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms + +--- +type: paper +title: "From Storage to Experience: A Survey on the Evolution of LLM Agent Memory Mechanisms" +authors: Jinghao Luo, Yuchen Tian, Chuxue Cao, Ziyang Luo, Hongzhan Lin, Kaixin Li, Chuyi Kong, Ruichao Yang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.06716 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-05-07 +status: queued +relevance: high +topics: + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.06716 diff --git a/papers/items/2026-2605-06737-a-self-healing-framework-for-reliable-llm-based-autonomous-agents.md b/papers/items/2026-2605-06737-a-self-healing-framework-for-reliable-llm-based-autonomous-agents.md new file mode 100644 index 0000000..dd167a7 --- /dev/null +++ b/papers/items/2026-2605-06737-a-self-healing-framework-for-reliable-llm-based-autonomous-agents.md @@ -0,0 +1,64 @@ +# Paper: A Self-Healing Framework for Reliable LLM-Based Autonomous Agents + +--- +type: paper +title: A Self-Healing Framework for Reliable LLM-Based Autonomous Agents +authors: Cheonsu Jeong, Younggun Shin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.06737 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-05-07 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - planning + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, computer-use, multi-agent, planning, reasoning, workflow-agent +- arXiv categories: cs.SE, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.06737 diff --git a/papers/items/2026-2605-06812-towards-security-auditable-llm-agents-a-unified-graph-representation.md b/papers/items/2026-2605-06812-towards-security-auditable-llm-agents-a-unified-graph-representation.md new file mode 100644 index 0000000..ee08d14 --- /dev/null +++ b/papers/items/2026-2605-06812-towards-security-auditable-llm-agents-a-unified-graph-representation.md @@ -0,0 +1,64 @@ +# Paper: Towards Security-Auditable LLM Agents: A Unified Graph Representation + +--- +type: paper +title: "Towards Security-Auditable LLM Agents: A Unified Graph Representation" +authors: Chaofan Li, Lyuye Zhang, Jintao Zhai, Siyue Feng, Xichun Yang, Huahao Wang, Shihan Dou, Yu Ji, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.06812 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-05-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, memory, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.06812 diff --git a/papers/items/2026-2605-06869-agentick-a-unified-benchmark-for-general-sequential-decision-making-agents.md b/papers/items/2026-2605-06869-agentick-a-unified-benchmark-for-general-sequential-decision-making-agents.md new file mode 100644 index 0000000..ae45339 --- /dev/null +++ b/papers/items/2026-2605-06869-agentick-a-unified-benchmark-for-general-sequential-decision-making-agents.md @@ -0,0 +1,64 @@ +# Paper: Agentick: A Unified Benchmark for General Sequential Decision-Making Agents + +--- +type: paper +title: "Agentick: A Unified Benchmark for General Sequential Decision-Making Agents" +authors: Roger Creus Castanyer, Pablo Samuel Castro, Glen Berseth +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.06869 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-05-12 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 23 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 23 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.06869 diff --git a/papers/items/2026-2605-06890-beyond-the-black-box-interpretability-of-agentic-ai-tool-use.md b/papers/items/2026-2605-06890-beyond-the-black-box-interpretability-of-agentic-ai-tool-use.md new file mode 100644 index 0000000..12685a9 --- /dev/null +++ b/papers/items/2026-2605-06890-beyond-the-black-box-interpretability-of-agentic-ai-tool-use.md @@ -0,0 +1,63 @@ +# Paper: Beyond the Black Box: Interpretability of Agentic AI Tool Use + +--- +type: paper +title: "Beyond the Black Box: Interpretability of Agentic AI Tool Use" +authors: Hariom Tatsat, Ariye Shater +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.06890 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, planning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.MA +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.06890 diff --git a/papers/items/2026-2605-06957-learning-and-reusing-policy-decompositions-for-hierarchical-generalized-planning.md b/papers/items/2026-2605-06957-learning-and-reusing-policy-decompositions-for-hierarchical-generalized-planning.md new file mode 100644 index 0000000..582cde8 --- /dev/null +++ b/papers/items/2026-2605-06957-learning-and-reusing-policy-decompositions-for-hierarchical-generalized-planning.md @@ -0,0 +1,60 @@ +# Paper: Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents + +--- +type: paper +title: Learning and Reusing Policy Decompositions for Hierarchical Generalized Planning with LLM Agents +authors: Shirin Sohrabi, Haritha Ananthakrishnan, Harsha Kokel, Kavitha Srinivas, Michael Katz +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.06957 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-05-07 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.06957 diff --git a/papers/items/2026-2605-06992-why-does-agentic-safety-fail-to-generalize-across-tasks.md b/papers/items/2026-2605-06992-why-does-agentic-safety-fail-to-generalize-across-tasks.md new file mode 100644 index 0000000..e2612a5 --- /dev/null +++ b/papers/items/2026-2605-06992-why-does-agentic-safety-fail-to-generalize-across-tasks.md @@ -0,0 +1,60 @@ +# Paper: Why Does Agentic Safety Fail to Generalize Across Tasks? + +--- +type: paper +title: Why Does Agentic Safety Fail to Generalize Across Tasks? +authors: Yonatan Slutzky, Yotam Alexander, Tomer Slor, Yoav Nagel, Nadav Cohen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.06992 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-07 +updated_at: 2026-05-07 +status: queued +relevance: high +topics: + - agent-safety + - embodied-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - stat.ML +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, embodied-agent +- arXiv categories: cs.LG, stat.ML +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.06992 diff --git a/papers/items/2026-2605-07112-switchcraft-ai-model-router-for-agentic-tool-calling.md b/papers/items/2026-2605-07112-switchcraft-ai-model-router-for-agentic-tool-calling.md new file mode 100644 index 0000000..982ba2e --- /dev/null +++ b/papers/items/2026-2605-07112-switchcraft-ai-model-router-for-agentic-tool-calling.md @@ -0,0 +1,61 @@ +# Paper: Switchcraft: AI Model Router for Agentic Tool Calling + +--- +type: paper +title: "Switchcraft: AI Model Router for Agentic Tool Calling" +authors: Sharad Agarwal, Pooria Namyar, Alec Wolman, Rahul Ambavat, Ankur Gupta, Qizheng Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.07112 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-08 +updated_at: 2026-05-08 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, reasoning, tool-use +- arXiv categories: cs.AI, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.07112 diff --git a/papers/items/2026-2605-07251-can-agents-price-a-reaction-evaluating-llms-on-chemical-cost-reasoning.md b/papers/items/2026-2605-07251-can-agents-price-a-reaction-evaluating-llms-on-chemical-cost-reasoning.md new file mode 100644 index 0000000..f406c1a --- /dev/null +++ b/papers/items/2026-2605-07251-can-agents-price-a-reaction-evaluating-llms-on-chemical-cost-reasoning.md @@ -0,0 +1,62 @@ +# Paper: Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning + +--- +type: paper +title: Can Agents Price a Reaction? Evaluating LLMs on Chemical Cost Reasoning +authors: Yuyang Wu, Yue Huang, Shuaike Shen, Xujian Wang, Shuhao Zhang, Qiyao Xue, Weichen Liu, Runtian Gao, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.07251 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-08 +updated_at: 2026-05-08 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.07251 diff --git a/papers/items/2026-2605-07830-cybiasbench-benchmarking-bias-in-llm-agents-for-cyber-attack-scenarios.md b/papers/items/2026-2605-07830-cybiasbench-benchmarking-bias-in-llm-agents-for-cyber-attack-scenarios.md new file mode 100644 index 0000000..1e66faa --- /dev/null +++ b/papers/items/2026-2605-07830-cybiasbench-benchmarking-bias-in-llm-agents-for-cyber-attack-scenarios.md @@ -0,0 +1,60 @@ +# Paper: CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios + +--- +type: paper +title: "CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios" +authors: Taein Lim, Seongyong Ju, Munhyeok Kim, Hyunjun Kim, Hoki Kim +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.07830 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-08 +updated_at: 2026-05-08 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety +- arXiv categories: cs.CR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.07830 diff --git a/papers/items/2026-2605-08374-memq-integrating-q-learning-into-self-evolving-memory-agents-over-provenance-dag.md b/papers/items/2026-2605-08374-memq-integrating-q-learning-into-self-evolving-memory-agents-over-provenance-dag.md new file mode 100644 index 0000000..44a2c77 --- /dev/null +++ b/papers/items/2026-2605-08374-memq-integrating-q-learning-into-self-evolving-memory-agents-over-provenance-dag.md @@ -0,0 +1,64 @@ +# Paper: MemQ: Integrating Q-Learning into Self-Evolving Memory Agents over Provenance DAGs + +--- +type: paper +title: "MemQ: Integrating Q-Learning into Self-Evolving Memory Agents over Provenance DAGs" +authors: Junwei Liao, Haoting Shi, Ruiwen Zhou, Jiaqian Wang, Shengtao Zhang, Wei Zhang, Ying Wen, Zhiyu Li, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.08374 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-08 +updated_at: 2026-05-14 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, computer-use, embodied-agent, memory, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.08374 diff --git a/papers/items/2026-2605-08442-defense-effectiveness-across-architectural-layers-a-mechanistic-evaluation-of-pe.md b/papers/items/2026-2605-08442-defense-effectiveness-across-architectural-layers-a-mechanistic-evaluation-of-pe.md new file mode 100644 index 0000000..ac8ead1 --- /dev/null +++ b/papers/items/2026-2605-08442-defense-effectiveness-across-architectural-layers-a-mechanistic-evaluation-of-pe.md @@ -0,0 +1,66 @@ +# Paper: Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents + +--- +type: paper +title: "Defense effectiveness across architectural layers: a mechanistic evaluation of persistent memory attacks on stateful LLM agents" +authors: Jun Wen Leong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.08442 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-08 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 23 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, rag, reasoning, tool-use +- arXiv categories: cs.CR, cs.AI, cs.LG +- collection score: 23 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.08442 diff --git a/papers/items/2026-2605-08763-when-llms-team-up-a-coordinated-attack-framework-for-automated-cyber-intrusions.md b/papers/items/2026-2605-08763-when-llms-team-up-a-coordinated-attack-framework-for-automated-cyber-intrusions.md new file mode 100644 index 0000000..bd2ce35 --- /dev/null +++ b/papers/items/2026-2605-08763-when-llms-team-up-a-coordinated-attack-framework-for-automated-cyber-intrusions.md @@ -0,0 +1,64 @@ +# Paper: When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions + +--- +type: paper +title: "When LLMs Team Up: A Coordinated Attack Framework for Automated Cyber Intrusions" +authors: Minfeng Qi, Tianqing Zhu, Zijie Xu, Congcong Zhu, Qin Wang, Wanlei Zhou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.08763 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-09 +updated_at: 2026-05-09 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, multi-agent, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.CR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.08763 diff --git a/papers/items/2026-2605-08876-otora-a-unified-red-teaming-framework-for-reasoning-level-denial-of-service-in-l.md b/papers/items/2026-2605-08876-otora-a-unified-red-teaming-framework-for-reasoning-level-denial-of-service-in-l.md new file mode 100644 index 0000000..78edad4 --- /dev/null +++ b/papers/items/2026-2605-08876-otora-a-unified-red-teaming-framework-for-reasoning-level-denial-of-service-in-l.md @@ -0,0 +1,60 @@ +# Paper: OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents + +--- +type: paper +title: "OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents" +authors: Xinyu Li, Ronghui Mu, Lin Li, Tianjin Huang, Gaojie Jin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.08876 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-09 +updated_at: 2026-06-07 +status: queued +relevance: high +topics: + - computer-use + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: computer-use, reasoning, tool-use +- arXiv categories: cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.08876 diff --git a/papers/items/2026-2605-08964-trustworthy-ai-ensuring-reliability-and-accountability-from-models-to-agents.md b/papers/items/2026-2605-08964-trustworthy-ai-ensuring-reliability-and-accountability-from-models-to-agents.md new file mode 100644 index 0000000..652faf1 --- /dev/null +++ b/papers/items/2026-2605-08964-trustworthy-ai-ensuring-reliability-and-accountability-from-models-to-agents.md @@ -0,0 +1,63 @@ +# Paper: Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents + +--- +type: paper +title: "Trustworthy AI: Ensuring Reliability and Accountability from Models to Agents" +authors: Carol Xuan Long +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.08964 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-09 +updated_at: 2026-05-09 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, coding-agent, multi-agent, rag, tool-use +- arXiv categories: cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.08964 diff --git a/papers/items/2026-2605-09168-civex-causal-intervention-verification-for-language-agents.md b/papers/items/2026-2605-09168-civex-causal-intervention-verification-for-language-agents.md new file mode 100644 index 0000000..6102580 --- /dev/null +++ b/papers/items/2026-2605-09168-civex-causal-intervention-verification-for-language-agents.md @@ -0,0 +1,62 @@ +# Paper: CIVeX: Causal Intervention Verification for Language Agents + +--- +type: paper +title: "CIVeX: Causal Intervention Verification for Language Agents" +authors: Fabio Rovai +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.09168 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-09 +updated_at: 2026-05-09 +status: queued +relevance: high +topics: + - agent-safety + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-safety, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.09168 diff --git a/papers/items/2026-2605-09692-causal-state-binding-predicts-action-control-in-language-agents.md b/papers/items/2026-2605-09692-causal-state-binding-predicts-action-control-in-language-agents.md new file mode 100644 index 0000000..b033172 --- /dev/null +++ b/papers/items/2026-2605-09692-causal-state-binding-predicts-action-control-in-language-agents.md @@ -0,0 +1,62 @@ +# Paper: Causal state binding predicts action control in language agents + +--- +type: paper +title: Causal state binding predicts action control in language agents +authors: Xiao Jia +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.09692 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-10 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, coding-agent, memory, planning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.09692 diff --git a/papers/items/2026-2605-10365-agent-valuebench-a-comprehensive-benchmark-for-evaluating-agent-values.md b/papers/items/2026-2605-10365-agent-valuebench-a-comprehensive-benchmark-for-evaluating-agent-values.md new file mode 100644 index 0000000..bf5c525 --- /dev/null +++ b/papers/items/2026-2605-10365-agent-valuebench-a-comprehensive-benchmark-for-evaluating-agent-values.md @@ -0,0 +1,60 @@ +# Paper: Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values + +--- +type: paper +title: "Agent-ValueBench: A Comprehensive Benchmark for Evaluating Agent Values" +authors: Haonan Dong, Qiguan Feng, Kehan Jiang, Haoran Ye, Xin Zhang, Guojie Song +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.10365 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-11 +updated_at: 2026-05-11 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.10365 diff --git a/papers/items/2026-2605-10763-matra-modeling-the-attack-surface-of-agentic-ai-systems-openclaw-case-study.md b/papers/items/2026-2605-10763-matra-modeling-the-attack-surface-of-agentic-ai-systems-openclaw-case-study.md new file mode 100644 index 0000000..c2076c4 --- /dev/null +++ b/papers/items/2026-2605-10763-matra-modeling-the-attack-surface-of-agentic-ai-systems-openclaw-case-study.md @@ -0,0 +1,62 @@ +# Paper: MATRA: Modeling the Attack Surface of Agentic AI Systems -- OpenClaw Case Study + +--- +type: paper +title: "MATRA: Modeling the Attack Surface of Agentic AI Systems -- OpenClaw Case Study" +authors: Tim Van hamme, Thomas Vissers, Javier Carnerero-Cano, Mario Fritz, Emil C. Lupu, Lieven Desmet, Dinil Mon Divakaran +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.10763 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-11 +updated_at: 2026-05-11 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.AI, cs.CR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.10763 diff --git a/papers/items/2026-2605-10779-litmus-benchmarking-behavioral-jailbreaks-of-llm-agents-in-real-os-environments.md b/papers/items/2026-2605-10779-litmus-benchmarking-behavioral-jailbreaks-of-llm-agents-in-real-os-environments.md new file mode 100644 index 0000000..88e23c6 --- /dev/null +++ b/papers/items/2026-2605-10779-litmus-benchmarking-behavioral-jailbreaks-of-llm-agents-in-real-os-environments.md @@ -0,0 +1,62 @@ +# Paper: LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments + +--- +type: paper +title: "LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments" +authors: Chiyu Zhang, Huiqin Yang, Bendong Jiang, Xiaolei Zhang, Yiran Zhao, Ruyi Chen, Lu Zhou, Xiaogang Xu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.10779 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-11 +updated_at: 2026-05-11 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, tool-use +- arXiv categories: cs.CR, cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.10779 diff --git a/papers/items/2026-2605-10870-remember-the-decision-not-the-description-a-rate-distortion-framework-for-agent-.md b/papers/items/2026-2605-10870-remember-the-decision-not-the-description-a-rate-distortion-framework-for-agent-.md new file mode 100644 index 0000000..e1dce16 --- /dev/null +++ b/papers/items/2026-2605-10870-remember-the-decision-not-the-description-a-rate-distortion-framework-for-agent-.md @@ -0,0 +1,60 @@ +# Paper: Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory + +--- +type: paper +title: "Remember the Decision, Not the Description: A Rate-Distortion Framework for Agent Memory" +authors: Mingxi Zou, Zhihan Guo, Langzhang Liang, Zhuo Wang, Qifan Wang, Qingsong Wen, Irwin King, Lizhen Qu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.10870 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-11 +updated_at: 2026-05-11 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, planning +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.10870 diff --git a/papers/items/2026-2605-11039-the-granularity-mismatch-in-agent-security-argument-level-provenance-solves-enfo.md b/papers/items/2026-2605-11039-the-granularity-mismatch-in-agent-security-argument-level-provenance-solves-enfo.md new file mode 100644 index 0000000..7133dbc --- /dev/null +++ b/papers/items/2026-2605-11039-the-granularity-mismatch-in-agent-security-argument-level-provenance-solves-enfo.md @@ -0,0 +1,65 @@ +# Paper: The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck + +--- +type: paper +title: "The Granularity Mismatch in Agent Security: Argument-Level Provenance Solves Enforcement and Isolates the LLM Reasoning Bottleneck" +authors: Linfeng Fan, Ziwei Li, Yuan Tian, Yichen Wang, Rongsheng Li, Xiong Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.11039 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-11 +updated_at: 2026-05-11 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.11039 diff --git a/papers/items/2026-2605-11225-pivot-bridging-planning-and-execution-in-llm-agents-via-trajectory-refinement.md b/papers/items/2026-2605-11225-pivot-bridging-planning-and-execution-in-llm-agents-via-trajectory-refinement.md new file mode 100644 index 0000000..d2a2745 --- /dev/null +++ b/papers/items/2026-2605-11225-pivot-bridging-planning-and-execution-in-llm-agents-via-trajectory-refinement.md @@ -0,0 +1,64 @@ +# Paper: PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement + +--- +type: paper +title: "PIVOT: Bridging Planning and Execution in LLM Agents via Trajectory Refinement" +authors: Tuo Zhang, Alin-Ionut Popa, Yan Xu, Rui Song, Dimitrios Dimitriadis +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.11225 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-11 +updated_at: 2026-05-11 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: autonomous-agent-llm, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, planning-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, planning, tool-use +- arXiv categories: cs.AI, cs.LG, cs.MA +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.11225 diff --git a/papers/items/2026-2605-11388-deep-reasoning-in-general-purpose-agents-via-structured-meta-cognition.md b/papers/items/2026-2605-11388-deep-reasoning-in-general-purpose-agents-via-structured-meta-cognition.md new file mode 100644 index 0000000..04a3ad9 --- /dev/null +++ b/papers/items/2026-2605-11388-deep-reasoning-in-general-purpose-agents-via-structured-meta-cognition.md @@ -0,0 +1,63 @@ +# Paper: Deep Reasoning in General Purpose Agents via Structured Meta-Cognition + +--- +type: paper +title: Deep Reasoning in General Purpose Agents via Structured Meta-Cognition +authors: Dean Light, Michael Theologitis, Kshitish Ghate, Shuyue Stella Li, Benjamin Newman, Chirag Shah, Aylin Caliskan, Pang Wei Koh, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.11388 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-12 +updated_at: 2026-05-12 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, planning, rag, reasoning +- arXiv categories: cs.CL, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.11388 diff --git a/papers/items/2026-2605-11534-prism-planning-and-reasoning-with-intent-in-simulated-embodied-environments.md b/papers/items/2026-2605-11534-prism-planning-and-reasoning-with-intent-in-simulated-embodied-environments.md new file mode 100644 index 0000000..ec11f56 --- /dev/null +++ b/papers/items/2026-2605-11534-prism-planning-and-reasoning-with-intent-in-simulated-embodied-environments.md @@ -0,0 +1,64 @@ +# Paper: PRISM: : Planning and Reasoning with Intent in Simulated Embodied Environments + +--- +type: paper +title: "PRISM: : Planning and Reasoning with Intent in Simulated Embodied Environments" +authors: Yunn Kang Lim, Pengzhan Sun, Ziyi Bai, Xun Xu, Angela Yao, Xulei Yang, Shijie Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.11534 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-12 +updated_at: 2026-05-12 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, embodied-agent, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.RO +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.11534 diff --git a/papers/items/2026-2605-11633-can-llm-agents-respond-to-disasters-benchmarking-heterogeneous-geospatial-reason.md b/papers/items/2026-2605-11633-can-llm-agents-respond-to-disasters-benchmarking-heterogeneous-geospatial-reason.md new file mode 100644 index 0000000..736ff21 --- /dev/null +++ b/papers/items/2026-2605-11633-can-llm-agents-respond-to-disasters-benchmarking-heterogeneous-geospatial-reason.md @@ -0,0 +1,63 @@ +# Paper: Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations + +--- +type: paper +title: Can LLM Agents Respond to Disasters? Benchmarking Heterogeneous Geospatial Reasoning in Emergency Operations +authors: Junjue Wang, Weihao Xuan, Heli Qi, Pengyu Dai, Kunyi Liu, Hongruixuan Chen, Zhuo Zheng, Junshi Xia, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.11633 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-12 +updated_at: 2026-05-12 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.11633 diff --git a/papers/items/2026-2605-11882-on-policy-self-evolution-via-failure-trajectories-for-agentic-safety-alignment.md b/papers/items/2026-2605-11882-on-policy-self-evolution-via-failure-trajectories-for-agentic-safety-alignment.md new file mode 100644 index 0000000..b5fc85b --- /dev/null +++ b/papers/items/2026-2605-11882-on-policy-self-evolution-via-failure-trajectories-for-agentic-safety-alignment.md @@ -0,0 +1,60 @@ +# Paper: On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment + +--- +type: paper +title: On-Policy Self-Evolution via Failure Trajectories for Agentic Safety Alignment +authors: Bo Yin, Qi Li, Xinchao Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.11882 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-12 +updated_at: 2026-05-12 +status: queued +relevance: high +topics: + - agent-safety + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.11882 diff --git a/papers/items/2026-2605-11928-when-simulation-lies-a-sim-to-real-benchmark-and-domain-randomized-rl-recipe-for.md b/papers/items/2026-2605-11928-when-simulation-lies-a-sim-to-real-benchmark-and-domain-randomized-rl-recipe-for.md new file mode 100644 index 0000000..44945d3 --- /dev/null +++ b/papers/items/2026-2605-11928-when-simulation-lies-a-sim-to-real-benchmark-and-domain-randomized-rl-recipe-for.md @@ -0,0 +1,60 @@ +# Paper: When Simulation Lies: A Sim-to-Real Benchmark and Domain-Randomized RL Recipe for Tool-Use Agents + +--- +type: paper +title: "When Simulation Lies: A Sim-to-Real Benchmark and Domain-Randomized RL Recipe for Tool-Use Agents" +authors: Xiaolin Zhou, Aojie Yuan, Zheng Luo, Zipeng Ling, Xixiao Pan, Yicheng Gao, Haiyue Zhang, Jiate Li, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.11928 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-12 +updated_at: 2026-05-12 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling, language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling, language-agent +- inferred topics: agent-evaluation, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.11928 diff --git a/papers/items/2026-2605-11946-counterfactual-trace-auditing-of-llm-agent-skills.md b/papers/items/2026-2605-11946-counterfactual-trace-auditing-of-llm-agent-skills.md new file mode 100644 index 0000000..b48f819 --- /dev/null +++ b/papers/items/2026-2605-11946-counterfactual-trace-auditing-of-llm-agent-skills.md @@ -0,0 +1,61 @@ +# Paper: Counterfactual Trace Auditing of LLM Agent Skills + +--- +type: paper +title: Counterfactual Trace Auditing of LLM Agent Skills +authors: Xiaolin Zhou, Jinbo Liu, Li Li, Ryan A. Rossi, Xiyang Hu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.11946 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-12 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, coding-agent, planning, rag +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.11946 diff --git a/papers/items/2026-2605-12015-skillsafetybench-evaluating-agent-safety-under-skill-facing-attack-surfaces.md b/papers/items/2026-2605-12015-skillsafetybench-evaluating-agent-safety-under-skill-facing-attack-surfaces.md new file mode 100644 index 0000000..bf9c70d --- /dev/null +++ b/papers/items/2026-2605-12015-skillsafetybench-evaluating-agent-safety-under-skill-facing-attack-surfaces.md @@ -0,0 +1,68 @@ +# Paper: SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces + +--- +type: paper +title: "SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces" +authors: Chang Jin, An Wang, Zeming Wei, Kai Wang, Biaojie Zeng, Qiaosheng Zhang, Chao Yang, Jingjing Qu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.12015 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-12 +updated_at: 2026-05-27 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.CL + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI, cs.CL, cs.LG, cs.MA +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.12015 diff --git a/papers/items/2026-2605-12061-sage-a-self-evolving-agentic-graph-memory-engine-for-structure-aware-associative.md b/papers/items/2026-2605-12061-sage-a-self-evolving-agentic-graph-memory-engine-for-structure-aware-associative.md new file mode 100644 index 0000000..4f8ed94 --- /dev/null +++ b/papers/items/2026-2605-12061-sage-a-self-evolving-agentic-graph-memory-engine-for-structure-aware-associative.md @@ -0,0 +1,62 @@ +# Paper: SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory + +--- +type: paper +title: "SAGE: A Self-Evolving Agentic Graph-Memory Engine for Structure-Aware Associative Memory" +authors: Juntong Wang, Haoyue Zhao, guanghui Pan, Xiyuan Wang, Yanbo Wang, Qiyan Deng, Muhan Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.12061 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-12 +updated_at: 2026-05-12 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.12061 diff --git a/papers/items/2026-2605-12260-prism-pareto-efficient-retrieval-over-intent-aware-structured-memory-for-long-ho.md b/papers/items/2026-2605-12260-prism-pareto-efficient-retrieval-over-intent-aware-structured-memory-for-long-ho.md new file mode 100644 index 0000000..05944ee --- /dev/null +++ b/papers/items/2026-2605-12260-prism-pareto-efficient-retrieval-over-intent-aware-structured-memory-for-long-ho.md @@ -0,0 +1,62 @@ +# Paper: PRISM: Pareto-Efficient Retrieval over Intent-Aware Structured Memory for Long-Horizon Agents + +--- +type: paper +title: "PRISM: Pareto-Efficient Retrieval over Intent-Aware Structured Memory for Long-Horizon Agents" +authors: Jingyi Peng, Zhongwei Wan, Weiting Liu, Qiuzhuang Sun +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.12260 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-12 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, planning, rag, tool-use +- arXiv categories: cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.12260 diff --git a/papers/items/2026-2605-13481-personalai-2-0-enhancing-knowledge-graph-traversal-retrieval-with-planning-mecha.md b/papers/items/2026-2605-13481-personalai-2-0-enhancing-knowledge-graph-traversal-retrieval-with-planning-mecha.md new file mode 100644 index 0000000..2023c24 --- /dev/null +++ b/papers/items/2026-2605-13481-personalai-2-0-enhancing-knowledge-graph-traversal-retrieval-with-planning-mecha.md @@ -0,0 +1,62 @@ +# Paper: PersonalAI 2.0: Enhancing knowledge graph traversal/retrieval with planning mechanism for Personalized LLM Agents + +--- +type: paper +title: "PersonalAI 2.0: Enhancing knowledge graph traversal/retrieval with planning mechanism for Personalized LLM Agents" +authors: Mikhail Menschikov, Matvey Iskornev, Alexander Kharitonov, Alina Bogdanova, Mikhail Belkin, Ekaterina Lisitsyna, Artyom Sosedka, Victoria Dochkina, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.13481 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-13 +updated_at: 2026-05-13 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, planning, rag, reasoning +- arXiv categories: cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.13481 diff --git a/papers/items/2026-2605-13542-realicu-do-llm-agents-understand-long-context-icu-data-a-benchmark-beyond-behavi.md b/papers/items/2026-2605-13542-realicu-do-llm-agents-understand-long-context-icu-data-a-benchmark-beyond-behavi.md new file mode 100644 index 0000000..2614017 --- /dev/null +++ b/papers/items/2026-2605-13542-realicu-do-llm-agents-understand-long-context-icu-data-a-benchmark-beyond-behavi.md @@ -0,0 +1,66 @@ +# Paper: RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation + +--- +type: paper +title: "RealICU: Do LLM Agents Understand Long-Context ICU Data? A Benchmark Beyond Behavior Imitation" +authors: Chengzhi Shen, Weixiang Shen, Tobias Susetzky, Chen, Chen, Jun Li, Yuyuan Liu, Xuepeng Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.13542 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-13 +updated_at: 2026-05-13 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, memory, planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL, cs.LG, cs.MA +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.13542 diff --git a/papers/items/2026-2605-13618-openaaas-an-open-agent-as-a-service-framework-for-distributed-materials-informat.md b/papers/items/2026-2605-13618-openaaas-an-open-agent-as-a-service-framework-for-distributed-materials-informat.md new file mode 100644 index 0000000..2c29cd0 --- /dev/null +++ b/papers/items/2026-2605-13618-openaaas-an-open-agent-as-a-service-framework-for-distributed-materials-informat.md @@ -0,0 +1,64 @@ +# Paper: OpenAaaS: An Open Agent-as-a-Service Framework for Distributed Materials-Informatics Research + +--- +type: paper +title: "OpenAaaS: An Open Agent-as-a-Service Framework for Distributed Materials-Informatics Research" +authors: Peng Kang, Bixuan Li, Xiaoya Huang, Shuo Shi, Weiqiao Zhou, Zhen Li, Yu Liu, Lei Zheng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.13618 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-13 +updated_at: 2026-05-13 +status: queued +relevance: high +topics: + - memory + - multi-agent + - planning + - rag + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cond-mat.mtrl-sci + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: memory, multi-agent, planning, rag, reasoning, workflow-agent +- arXiv categories: cond-mat.mtrl-sci, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.13618 diff --git a/papers/items/2026-2605-13716-skillops-managing-llm-agent-skill-libraries-as-self-maintaining-software-ecosyst.md b/papers/items/2026-2605-13716-skillops-managing-llm-agent-skill-libraries-as-self-maintaining-software-ecosyst.md new file mode 100644 index 0000000..c86bf2f --- /dev/null +++ b/papers/items/2026-2605-13716-skillops-managing-llm-agent-skill-libraries-as-self-maintaining-software-ecosyst.md @@ -0,0 +1,62 @@ +# Paper: SkillOps: Managing LLM Agent Skill Libraries as Self-Maintaining Software Ecosystems + +--- +type: paper +title: "SkillOps: Managing LLM Agent Skill Libraries as Self-Maintaining Software Ecosystems" +authors: Hongji Pu, Xinyuan Song, Liang Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.13716 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-13 +updated_at: 2026-05-13 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, rag +- arXiv categories: cs.SE, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.13716 diff --git a/papers/items/2026-2605-14126-reinforcement-learning-for-tool-calling-agents-in-fast-healthcare-interoperabili.md b/papers/items/2026-2605-14126-reinforcement-learning-for-tool-calling-agents-in-fast-healthcare-interoperabili.md new file mode 100644 index 0000000..2a2d025 --- /dev/null +++ b/papers/items/2026-2605-14126-reinforcement-learning-for-tool-calling-agents-in-fast-healthcare-interoperabili.md @@ -0,0 +1,63 @@ +# Paper: Reinforcement Learning for Tool-Calling Agents in Fast Healthcare Interoperability Resources (FHIR) + +--- +type: paper +title: Reinforcement Learning for Tool-Calling Agents in Fast Healthcare Interoperability Resources (FHIR) +authors: Marius S. Knorr, Robert Müller, Jan P. Bremer, Nils Schweingruber +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14126 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-13 +updated_at: 2026-05-13 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use +- arXiv categories: cs.LG, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14126 diff --git a/papers/items/2026-2605-14290-web-agents-should-adopt-the-plan-then-execute-paradigm.md b/papers/items/2026-2605-14290-web-agents-should-adopt-the-plan-then-execute-paradigm.md new file mode 100644 index 0000000..71b4521 --- /dev/null +++ b/papers/items/2026-2605-14290-web-agents-should-adopt-the-plan-then-execute-paradigm.md @@ -0,0 +1,65 @@ +# Paper: Web Agents Should Adopt the Plan-Then-Execute Paradigm + +--- +type: paper +title: Web Agents Should Adopt the Plan-Then-Execute Paradigm +authors: Julien Piet, Annabella Chow, Yiwei Hou, Muxi Lyu, Sylvie Venuto, Jinhao Zhu, Raluca Ada Popa, David Wagner +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14290 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-14 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.CL + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, planning, tool-use +- arXiv categories: cs.CR, cs.AI, cs.CL, cs.SE +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14290 diff --git a/papers/items/2026-2605-14322-are-agents-ready-to-teach-a-multi-stage-benchmark-for-real-world-teaching-workfl.md b/papers/items/2026-2605-14322-are-agents-ready-to-teach-a-multi-stage-benchmark-for-real-world-teaching-workfl.md new file mode 100644 index 0000000..bccc2fc --- /dev/null +++ b/papers/items/2026-2605-14322-are-agents-ready-to-teach-a-multi-stage-benchmark-for-real-world-teaching-workfl.md @@ -0,0 +1,60 @@ +# Paper: Are Agents Ready to Teach? A Multi-Stage Benchmark for Real-World Teaching Workflows + +--- +type: paper +title: Are Agents Ready to Teach? A Multi-Stage Benchmark for Real-World Teaching Workflows +authors: Zixin Chen, Peng Liu, Rui Sheng, Haobo Li, Jianhong Tu, Xiaodong Deng, Kashun Shum, Dayiheng Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14322 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-20 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14322 diff --git a/papers/items/2026-2605-14421-memlineage-lineage-guided-enforcement-for-llm-agent-memory.md b/papers/items/2026-2605-14421-memlineage-lineage-guided-enforcement-for-llm-agent-memory.md new file mode 100644 index 0000000..618a86b --- /dev/null +++ b/papers/items/2026-2605-14421-memlineage-lineage-guided-enforcement-for-llm-agent-memory.md @@ -0,0 +1,62 @@ +# Paper: MemLineage: Lineage-Guided Enforcement for LLM Agent Memory + +--- +type: paper +title: "MemLineage: Lineage-Guided Enforcement for LLM Agent Memory" +authors: Ciyan Ouyang, Rui Hou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14421 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-14 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14421 diff --git a/papers/items/2026-2605-14460-exploiting-llm-agent-supply-chains-via-payload-less-skills.md b/papers/items/2026-2605-14460-exploiting-llm-agent-supply-chains-via-payload-less-skills.md new file mode 100644 index 0000000..f5a4d2f --- /dev/null +++ b/papers/items/2026-2605-14460-exploiting-llm-agent-supply-chains-via-payload-less-skills.md @@ -0,0 +1,62 @@ +# Paper: Exploiting LLM Agent Supply Chains via Payload-less Skills + +--- +type: paper +title: Exploiting LLM Agent Supply Chains via Payload-less Skills +authors: Xinyu Liu, Yukai Zhao, Xing Hu, Xin Xia +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14460 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-14 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.CR, cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14460 diff --git a/papers/items/2026-2605-14498-groupmembench-benchmarking-llm-agent-memory-in-multi-party-conversations.md b/papers/items/2026-2605-14498-groupmembench-benchmarking-llm-agent-memory-in-multi-party-conversations.md new file mode 100644 index 0000000..b3283f4 --- /dev/null +++ b/papers/items/2026-2605-14498-groupmembench-benchmarking-llm-agent-memory-in-multi-party-conversations.md @@ -0,0 +1,62 @@ +# Paper: GroupMemBench: Benchmarking LLM Agent Memory in Multi-Party Conversations + +--- +type: paper +title: "GroupMemBench: Benchmarking LLM Agent Memory in Multi-Party Conversations" +authors: Jingbo Yang, Kwei-Herng Lai, Xiaowen Wang, Shiyu Chang, Yaar Harari, Evgeniy Gabrilovich +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14498 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-16 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, rag, reasoning +- arXiv categories: cs.CL +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14498 diff --git a/papers/items/2026-2605-14527-lang2mlip-end-to-end-language-to-machine-learning-interatomic-potential-developm.md b/papers/items/2026-2605-14527-lang2mlip-end-to-end-language-to-machine-learning-interatomic-potential-developm.md new file mode 100644 index 0000000..da57fe0 --- /dev/null +++ b/papers/items/2026-2605-14527-lang2mlip-end-to-end-language-to-machine-learning-interatomic-potential-developm.md @@ -0,0 +1,64 @@ +# Paper: Lang2MLIP: End-to-End Language-to-Machine Learning Interatomic Potential Development with Autonomous Agentic Workflows + +--- +type: paper +title: "Lang2MLIP: End-to-End Language-to-Machine Learning Interatomic Potential Development with Autonomous Agentic Workflows" +authors: Wenwen Li, Yuki Orimo, Nontawat Charoenphakdee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14527 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-14 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cond-mat.mtrl-sci + - physics.comp-ph +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, multi-agent, tool-use, workflow-agent, world-model +- arXiv categories: cs.LG, cond-mat.mtrl-sci, physics.comp-ph +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14527 diff --git a/papers/items/2026-2605-14892-beyond-individual-intelligence-surveying-collaboration-failure-attribution-and-s.md b/papers/items/2026-2605-14892-beyond-individual-intelligence-surveying-collaboration-failure-attribution-and-s.md new file mode 100644 index 0000000..ad1ee88 --- /dev/null +++ b/papers/items/2026-2605-14892-beyond-individual-intelligence-surveying-collaboration-failure-attribution-and-s.md @@ -0,0 +1,63 @@ +# Paper: Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems + +--- +type: paper +title: "Beyond Individual Intelligence: Surveying Collaboration, Failure Attribution, and Self-Evolution in LLM-based Multi-Agent Systems" +authors: Shihao Qi, Jie Ma, Rui Xing, Wei Guo, Xiao Huang, Zhitao Gao, Jianhao Deng, Jun Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14892 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-15 +status: queued +relevance: high +topics: + - agent-safety + - multi-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-safety, multi-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14892 diff --git a/papers/items/2026-2605-14906-memlens-benchmarking-multimodal-long-term-memory-in-large-vision-language-models.md b/papers/items/2026-2605-14906-memlens-benchmarking-multimodal-long-term-memory-in-large-vision-language-models.md new file mode 100644 index 0000000..e00145d --- /dev/null +++ b/papers/items/2026-2605-14906-memlens-benchmarking-multimodal-long-term-memory-in-large-vision-language-models.md @@ -0,0 +1,62 @@ +# Paper: MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models + +--- +type: paper +title: "MemLens: Benchmarking Multimodal Long-Term Memory in Large Vision-Language Models" +authors: Xiyu Ren, Zhaowei Wang, Yiming Du, Zhongwei Xie, Chi Liu, Xinlin Yang, Haoyue Feng, Wenjun Pan, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14906 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-14 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.CV +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14906 diff --git a/papers/items/2026-2605-14932-toward-securing-ai-agents-like-operating-systems.md b/papers/items/2026-2605-14932-toward-securing-ai-agents-like-operating-systems.md new file mode 100644 index 0000000..ddae9c5 --- /dev/null +++ b/papers/items/2026-2605-14932-toward-securing-ai-agents-like-operating-systems.md @@ -0,0 +1,61 @@ +# Paper: Toward Securing AI Agents Like Operating Systems + +--- +type: paper +title: Toward Securing AI Agents Like Operating Systems +authors: Lukas Pirch, Micha Horlboge, Patrick Großmann, Syeda Mahnur Asif, Klim Kireev, Thorsten Holz, Konrad Rieck +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.14932 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-14 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, computer-use, tool-use +- arXiv categories: cs.CR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.14932 diff --git a/papers/items/2026-2605-15040-orchard-an-open-source-agentic-modeling-framework.md b/papers/items/2026-2605-15040-orchard-an-open-source-agentic-modeling-framework.md new file mode 100644 index 0000000..1f1c5d0 --- /dev/null +++ b/papers/items/2026-2605-15040-orchard-an-open-source-agentic-modeling-framework.md @@ -0,0 +1,64 @@ +# Paper: Orchard: An Open-Source Agentic Modeling Framework + +--- +type: paper +title: "Orchard: An Open-Source Agentic Modeling Framework" +authors: Baolin Peng, Wenlin Yao, Qianhui Wu, Hao Cheng, Xiao Yu, Rui Yang, Tao Ge, Alessandro Sordoni, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.15040 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-21 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, coding-agent, computer-use, planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.15040 diff --git a/papers/items/2026-2605-15128-memeye-a-visual-centric-evaluation-framework-for-multimodal-agent-memory.md b/papers/items/2026-2605-15128-memeye-a-visual-centric-evaluation-framework-for-multimodal-agent-memory.md new file mode 100644 index 0000000..4106b53 --- /dev/null +++ b/papers/items/2026-2605-15128-memeye-a-visual-centric-evaluation-framework-for-multimodal-agent-memory.md @@ -0,0 +1,63 @@ +# Paper: MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory + +--- +type: paper +title: "MemEye: A Visual-Centric Evaluation Framework for Multimodal Agent Memory" +authors: Minghao Guo, Qingyue Jiao, Zeru Shi, Yihao Quan, Boxuan Zhang, Danrui Li, Liwei Che, Wujiang Xu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.15128 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-14 +updated_at: 2026-05-14 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.CL + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, reasoning, tool-use +- arXiv categories: cs.CV, cs.CL, cs.IR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.15128 diff --git a/papers/items/2026-2605-15206-agentstop-terminating-local-ai-agents-early-to-save-energy-in-consumer-devices.md b/papers/items/2026-2605-15206-agentstop-terminating-local-ai-agents-early-to-save-energy-in-consumer-devices.md new file mode 100644 index 0000000..3cc7f8c --- /dev/null +++ b/papers/items/2026-2605-15206-agentstop-terminating-local-ai-agents-early-to-save-energy-in-consumer-devices.md @@ -0,0 +1,65 @@ +# Paper: AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices + +--- +type: paper +title: "AgentStop: Terminating Local AI Agents Early to Save Energy in Consumer Devices" +authors: Dzung Pham, Kleomenis Katevas, Ali Shahin Shamsabadi, Hamed Haddadi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.15206 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-01 +updated_at: 2026-05-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.DC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, coding-agent, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.LG, cs.AI, cs.DC +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.15206 diff --git a/papers/items/2026-2605-15625-colpackagent-agent-skill-guided-hard-particle-monte-carlo-workflows-for-colloida.md b/papers/items/2026-2605-15625-colpackagent-agent-skill-guided-hard-particle-monte-carlo-workflows-for-colloida.md new file mode 100644 index 0000000..82b9bee --- /dev/null +++ b/papers/items/2026-2605-15625-colpackagent-agent-skill-guided-hard-particle-monte-carlo-workflows-for-colloida.md @@ -0,0 +1,64 @@ +# Paper: ColPackAgent: Agent-Skill-Guided Hard-Particle Monte Carlo Workflows for Colloidal Packing + +--- +type: paper +title: "ColPackAgent: Agent-Skill-Guided Hard-Particle Monte Carlo Workflows for Colloidal Packing" +authors: Lijie Ding, Changwoo Do +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.15625 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-15 +updated_at: 2026-05-15 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cond-mat.soft +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use, workflow-agent, world-model +- arXiv categories: cs.AI, cond-mat.soft +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.15625 diff --git a/papers/items/2026-2605-15701-h-mem-a-novel-memory-mechanism-for-evolving-and-retrieving-agent-memory-via-a-hy.md b/papers/items/2026-2605-15701-h-mem-a-novel-memory-mechanism-for-evolving-and-retrieving-agent-memory-via-a-hy.md new file mode 100644 index 0000000..9c96b08 --- /dev/null +++ b/papers/items/2026-2605-15701-h-mem-a-novel-memory-mechanism-for-evolving-and-retrieving-agent-memory-via-a-hy.md @@ -0,0 +1,61 @@ +# Paper: H-Mem: A Novel Memory Mechanism for Evolving and Retrieving Agent Memory via a Hybrid Structure + +--- +type: paper +title: "H-Mem: A Novel Memory Mechanism for Evolving and Retrieving Agent Memory via a Hybrid Structure" +authors: Jiawei Yu, Yixiang Fang, Xilin Liu, Yuchi Ma +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.15701 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-15 +updated_at: 2026-05-15 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag +- arXiv categories: cs.CL, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.15701 diff --git a/papers/items/2026-2605-15710-smmbench-a-benchmark-for-source-distributed-multimodal-agent-memory.md b/papers/items/2026-2605-15710-smmbench-a-benchmark-for-source-distributed-multimodal-agent-memory.md new file mode 100644 index 0000000..dcc3809 --- /dev/null +++ b/papers/items/2026-2605-15710-smmbench-a-benchmark-for-source-distributed-multimodal-agent-memory.md @@ -0,0 +1,62 @@ +# Paper: SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory + +--- +type: paper +title: "SMMBench: A Benchmark for Source-Distributed Multimodal Agent Memory" +authors: Huacan Chai, Yukai Wang, Yingxuan Yang, Dan Peng, Yuanyi Song, Zhihui Fu, Weiwen Liu, Jianghao Lin, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.15710 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-15 +updated_at: 2026-05-15 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.15710 diff --git a/papers/items/2026-2605-15759-dimmem-dimensional-structuring-for-efficient-long-term-agent-memory.md b/papers/items/2026-2605-15759-dimmem-dimensional-structuring-for-efficient-long-term-agent-memory.md new file mode 100644 index 0000000..ee3ee42 --- /dev/null +++ b/papers/items/2026-2605-15759-dimmem-dimensional-structuring-for-efficient-long-term-agent-memory.md @@ -0,0 +1,61 @@ +# Paper: DimMem: Dimensional Structuring for Efficient Long-Term Agent Memory + +--- +type: paper +title: "DimMem: Dimensional Structuring for Efficient Long-Term Agent Memory" +authors: Wentao Qiu, Haotian Hu, Fanyi Wang, Jinwei Kong, Yu Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.15759 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-15 +updated_at: 2026-05-24 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.15759 diff --git a/papers/items/2026-2605-16233-forge-self-evolving-agent-memory-with-no-weight-updates-via-population-broadcast.md b/papers/items/2026-2605-16233-forge-self-evolving-agent-memory-with-no-weight-updates-via-population-broadcast.md new file mode 100644 index 0000000..58deff7 --- /dev/null +++ b/papers/items/2026-2605-16233-forge-self-evolving-agent-memory-with-no-weight-updates-via-population-broadcast.md @@ -0,0 +1,65 @@ +# Paper: FORGE: Self-Evolving Agent Memory With No Weight Updates via Population Broadcast + +--- +type: paper +title: "FORGE: Self-Evolving Agent Memory With No Weight Updates via Population Broadcast" +authors: Igor Bogdanov, Chung-Horng Lung, Thomas Kunz, Jie Gao, Adrian Taylor, Marzia Zaman +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.16233 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-15 +updated_at: 2026-05-15 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.LG + - cs.MA + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, reasoning +- arXiv categories: cs.AI, cs.CL, cs.LG, cs.MA, eess.SY +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.16233 diff --git a/papers/items/2026-2605-16481-visual-agentic-memory-enabling-online-long-video-understanding-via-online-indexi.md b/papers/items/2026-2605-16481-visual-agentic-memory-enabling-online-long-video-understanding-via-online-indexi.md new file mode 100644 index 0000000..09beba6 --- /dev/null +++ b/papers/items/2026-2605-16481-visual-agentic-memory-enabling-online-long-video-understanding-via-online-indexi.md @@ -0,0 +1,63 @@ +# Paper: Visual Agentic Memory: Enabling Online Long Video Understanding via Online Indexing, Hierarchical Memory, and Agentic Retrieval + +--- +type: paper +title: "Visual Agentic Memory: Enabling Online Long Video Understanding via Online Indexing, Hierarchical Memory, and Agentic Retrieval" +authors: Aiden Yiliu Li, Nels Numan, Anthony Steed +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.16481 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-15 +updated_at: 2026-05-15 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag, reasoning +- arXiv categories: cs.CV, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.16481 diff --git a/papers/items/2026-2605-16821-multi-paradigm-agent-interaction-in-practice-a-systematic-analysis-of-generator-.md b/papers/items/2026-2605-16821-multi-paradigm-agent-interaction-in-practice-a-systematic-analysis-of-generator-.md new file mode 100644 index 0000000..024ca54 --- /dev/null +++ b/papers/items/2026-2605-16821-multi-paradigm-agent-interaction-in-practice-a-systematic-analysis-of-generator-.md @@ -0,0 +1,63 @@ +# Paper: Multi-Paradigm Agent Interaction in Practice:A Systematic Analysis of Generator-Evaluator, ReAct Loop,and Adversarial Evaluation in the buddyMe Framework + +--- +type: paper +title: "Multi-Paradigm Agent Interaction in Practice:A Systematic Analysis of Generator-Evaluator, ReAct Loop,and Adversarial Evaluation in the buddyMe Framework" +authors: Xiaohua Wang, Chao Han, Kai Yu, XiaoLiang Xu, Liang Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.16821 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-16 +updated_at: 2026-05-16 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - multi-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, memory, multi-agent, planning, tool-use +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.16821 diff --git a/papers/items/2026-2605-17075-a-red-teaming-framework-for-evaluating-robustness-of-ai-enabled-security-orchest.md b/papers/items/2026-2605-17075-a-red-teaming-framework-for-evaluating-robustness-of-ai-enabled-security-orchest.md new file mode 100644 index 0000000..e463db9 --- /dev/null +++ b/papers/items/2026-2605-17075-a-red-teaming-framework-for-evaluating-robustness-of-ai-enabled-security-orchest.md @@ -0,0 +1,63 @@ +# Paper: A Red Teaming Framework for Evaluating Robustness of AI-enabled Security Orchestration, Automation, and Response Systems + +--- +type: paper +title: A Red Teaming Framework for Evaluating Robustness of AI-enabled Security Orchestration, Automation, and Response Systems +authors: Ayan Javeed Shaikh, Nathaniel D. Bastian, Ankit Shah +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.17075 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-16 +updated_at: 2026-05-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, planning, tool-use, workflow-agent, world-model +- arXiv categories: cs.CR +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.17075 diff --git a/papers/items/2026-2605-17348-taming-zombie-agents-a-markov-state-aware-framework-for-resilient-multi-agent-ev.md b/papers/items/2026-2605-17348-taming-zombie-agents-a-markov-state-aware-framework-for-resilient-multi-agent-ev.md new file mode 100644 index 0000000..5505ded --- /dev/null +++ b/papers/items/2026-2605-17348-taming-zombie-agents-a-markov-state-aware-framework-for-resilient-multi-agent-ev.md @@ -0,0 +1,61 @@ +# Paper: Taming "Zombie'' Agents: A Markov State-Aware Framework for Resilient Multi-Agent Evolution + +--- +type: paper +title: "Taming \"Zombie'' Agents: A Markov State-Aware Framework for Resilient Multi-Agent Evolution" +authors: Taolin Zhang, Pukun Zhao, Qizhou Chen, Jiuheng Wan, Chen Chen, Xiaofeng He, Chengyu Wang, Richang Hong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.17348 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-17 +updated_at: 2026-05-17 +status: queued +relevance: high +topics: + - agent-safety + - memory + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-safety, memory, multi-agent, reasoning +- arXiv categories: cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.17348 diff --git a/papers/items/2026-2605-17453-trust-no-tool-evaluating-and-defending-llm-agents-under-untrusted-tool-feedback.md b/papers/items/2026-2605-17453-trust-no-tool-evaluating-and-defending-llm-agents-under-untrusted-tool-feedback.md new file mode 100644 index 0000000..57dc52d --- /dev/null +++ b/papers/items/2026-2605-17453-trust-no-tool-evaluating-and-defending-llm-agents-under-untrusted-tool-feedback.md @@ -0,0 +1,61 @@ +# Paper: Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback + +--- +type: paper +title: "Trust No Tool: Evaluating and Defending LLM Agents under Untrusted Tool Feedback" +authors: Lecheng Yan, Ruizhe Li, Xicheng Han, Wenxi Li, Binwu Wang, Longyue Wang, Chenyang Lyu, Guanhua Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.17453 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-17 +updated_at: 2026-05-17 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CR, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.17453 diff --git a/papers/items/2026-2605-17625-episodic-semantic-memory-architecture-for-long-horizon-scientific-agents.md b/papers/items/2026-2605-17625-episodic-semantic-memory-architecture-for-long-horizon-scientific-agents.md new file mode 100644 index 0000000..4789f2f --- /dev/null +++ b/papers/items/2026-2605-17625-episodic-semantic-memory-architecture-for-long-horizon-scientific-agents.md @@ -0,0 +1,64 @@ +# Paper: Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents + +--- +type: paper +title: Episodic-Semantic Memory Architecture for Long-Horizon Scientific Agents +authors: Nikola Milosevic +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.17625 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-17 +updated_at: 2026-05-17 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.17625 diff --git a/papers/items/2026-2605-18284-commitdistill-a-lightweight-knowledge-centric-memory-layer-for-software-reposito.md b/papers/items/2026-2605-18284-commitdistill-a-lightweight-knowledge-centric-memory-layer-for-software-reposito.md new file mode 100644 index 0000000..da3a429 --- /dev/null +++ b/papers/items/2026-2605-18284-commitdistill-a-lightweight-knowledge-centric-memory-layer-for-software-reposito.md @@ -0,0 +1,64 @@ +# Paper: CommitDistill: A Lightweight Knowledge-Centric Memory Layer for Software Repositories + +--- +type: paper +title: "CommitDistill: A Lightweight Knowledge-Centric Memory Layer for Software Repositories" +authors: Divya Chukkapalli, Thejesh Avula, Aditya Aggarwal, Harsimran Singh, Amith Tallanki +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.18284 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-18 +updated_at: 2026-05-18 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, coding-agent, computer-use, memory, rag, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.18284 diff --git a/papers/items/2026-2605-18502-the-distance-based-formation-controller-design-for-multi-agent-systems-in-port-h.md b/papers/items/2026-2605-18502-the-distance-based-formation-controller-design-for-multi-agent-systems-in-port-h.md new file mode 100644 index 0000000..90d9c7b --- /dev/null +++ b/papers/items/2026-2605-18502-the-distance-based-formation-controller-design-for-multi-agent-systems-in-port-h.md @@ -0,0 +1,62 @@ +# Paper: The distance-based formation controller design for multi-agent systems in port-Hamiltonian form + +--- +type: paper +title: The distance-based formation controller design for multi-agent systems in port-Hamiltonian form +authors: Jingyi Zhao, Yongxin Wu, Héctor García de Marina, Yuhu Wu, Yann Le Gorrec +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.18502 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-18 +updated_at: 2026-05-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - math.OC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, multi-agent, tool-use, world-model +- arXiv categories: math.OC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.18502 diff --git a/papers/items/2026-2605-18652-mementogui-learning-agentic-multimodal-memory-control-for-long-horizon-gui-agent.md b/papers/items/2026-2605-18652-mementogui-learning-agentic-multimodal-memory-control-for-long-horizon-gui-agent.md new file mode 100644 index 0000000..2e342f0 --- /dev/null +++ b/papers/items/2026-2605-18652-mementogui-learning-agentic-multimodal-memory-control-for-long-horizon-gui-agent.md @@ -0,0 +1,63 @@ +# Paper: MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents + +--- +type: paper +title: "MementoGUI: Learning Agentic Multimodal Memory Control for Long-Horizon GUI Agents" +authors: Ziyun Zeng, Hang Hua, Bocheng Zou, Mu Cai, Rogerio Feris, Jiebo Luo +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.18652 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-18 +updated_at: 2026-05-18 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, planning, rag, tool-use +- arXiv categories: cs.CV +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.18652 diff --git a/papers/items/2026-2605-18672-position-a-three-layer-probabilistic-assume-guarantee-architecture-is-structural.md b/papers/items/2026-2605-18672-position-a-three-layer-probabilistic-assume-guarantee-architecture-is-structural.md new file mode 100644 index 0000000..35e89a7 --- /dev/null +++ b/papers/items/2026-2605-18672-position-a-three-layer-probabilistic-assume-guarantee-architecture-is-structural.md @@ -0,0 +1,60 @@ +# Paper: Position: A Three-Layer Probabilistic Assume-Guarantee Architecture Is Structurally Required for Safe LLM Agent Deployment + +--- +type: paper +title: "Position: A Three-Layer Probabilistic Assume-Guarantee Architecture Is Structurally Required for Safe LLM Agent Deployment" +authors: S. Bensalem, Y. Dong, M. Franzle, X. Huang, J. Kroger, D. Nickovic, A. Nouri, R. Roy, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.18672 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-18 +updated_at: 2026-05-18 +status: queued +relevance: high +topics: + - agent-safety + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, multi-agent, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.18672 diff --git a/papers/items/2026-2605-18930-oep-poisoning-self-evolving-llm-agents-via-locally-correct-but-non-transferable-.md b/papers/items/2026-2605-18930-oep-poisoning-self-evolving-llm-agents-via-locally-correct-but-non-transferable-.md new file mode 100644 index 0000000..be25b2a --- /dev/null +++ b/papers/items/2026-2605-18930-oep-poisoning-self-evolving-llm-agents-via-locally-correct-but-non-transferable-.md @@ -0,0 +1,63 @@ +# Paper: OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences + +--- +type: paper +title: "OEP: Poisoning Self-Evolving LLM Agents via Locally Correct but Non-Transferable Experiences" +authors: Kaixiang Wang, Jiong Lou, Zhaojiacheng Zhou, Jie Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.18930 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-18 +updated_at: 2026-05-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, memory, reasoning +- arXiv categories: cs.CR, cs.AI, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.18930 diff --git a/papers/items/2026-2605-19604-formal-skill-programmable-runtime-skills-for-efficient-and-accurate-llm-agents.md b/papers/items/2026-2605-19604-formal-skill-programmable-runtime-skills-for-efficient-and-accurate-llm-agents.md new file mode 100644 index 0000000..05c2627 --- /dev/null +++ b/papers/items/2026-2605-19604-formal-skill-programmable-runtime-skills-for-efficient-and-accurate-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: Formal Skill: Programmable Runtime Skills for Efficient and Accurate LLM Agents + +--- +type: paper +title: "Formal Skill: Programmable Runtime Skills for Efficient and Accurate LLM Agents" +authors: Xi Zhang, Meijun Gao, Yuntian Zhao, Xinyu Tan, Yilun Yao, Feiyu Wang, Yanshu Wang, Dingsiyi, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.19604 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-19 +updated_at: 2026-05-19 +status: queued +relevance: high +topics: + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.19604 diff --git a/papers/items/2026-2605-19952-rethinking-how-to-remember-beyond-atomic-facts-in-lifelong-llm-agent-memory.md b/papers/items/2026-2605-19952-rethinking-how-to-remember-beyond-atomic-facts-in-lifelong-llm-agent-memory.md new file mode 100644 index 0000000..426bb31 --- /dev/null +++ b/papers/items/2026-2605-19952-rethinking-how-to-remember-beyond-atomic-facts-in-lifelong-llm-agent-memory.md @@ -0,0 +1,62 @@ +# Paper: Rethinking How to Remember: Beyond Atomic Facts in Lifelong LLM Agent Memory + +--- +type: paper +title: "Rethinking How to Remember: Beyond Atomic Facts in Lifelong LLM Agent Memory" +authors: Jingwei Sun, Jianing Zhu, Jiangchao Yao, Tongliang Liu, Bo Han +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.19952 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-19 +updated_at: 2026-05-19 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.19952 diff --git a/papers/items/2026-2605-20306-wildroadbench-a-wild-aerial-road-damage-grounding-benchmark-for-vision-language-.md b/papers/items/2026-2605-20306-wildroadbench-a-wild-aerial-road-damage-grounding-benchmark-for-vision-language-.md new file mode 100644 index 0000000..71e1345 --- /dev/null +++ b/papers/items/2026-2605-20306-wildroadbench-a-wild-aerial-road-damage-grounding-benchmark-for-vision-language-.md @@ -0,0 +1,63 @@ +# Paper: WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents + +--- +type: paper +title: "WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents" +authors: Bingnan Liu, Chenhang Cui, Rui Huang, Jiani Luo, Zhirong Shen, Tinghao Wang, Xiande Huang, Lingbei Meng, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.20306 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-19 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, coding-agent, rag, reasoning, tool-use +- arXiv categories: cs.CV, cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.20306 diff --git a/papers/items/2026-2605-20315-mix-quant-quantized-prefilling-precise-decoding-for-agentic-llms.md b/papers/items/2026-2605-20315-mix-quant-quantized-prefilling-precise-decoding-for-agentic-llms.md new file mode 100644 index 0000000..378455e --- /dev/null +++ b/papers/items/2026-2605-20315-mix-quant-quantized-prefilling-precise-decoding-for-agentic-llms.md @@ -0,0 +1,64 @@ +# Paper: Mix-Quant: Quantized Prefilling, Precise Decoding for Agentic LLMs + +--- +type: paper +title: "Mix-Quant: Quantized Prefilling, Precise Decoding for Agentic LLMs" +authors: Haiquan Lu, Zigeng Chen, Gongfan Fang, Xinyin Ma, Xinchao Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.20315 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-19 +updated_at: 2026-05-19 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, coding-agent, memory, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.20315 diff --git a/papers/items/2026-2605-20616-auto-dreamer-learning-offline-memory-consolidation-for-language-agents.md b/papers/items/2026-2605-20616-auto-dreamer-learning-offline-memory-consolidation-for-language-agents.md new file mode 100644 index 0000000..cf3e852 --- /dev/null +++ b/papers/items/2026-2605-20616-auto-dreamer-learning-offline-memory-consolidation-for-language-agents.md @@ -0,0 +1,61 @@ +# Paper: Auto-Dreamer: Learning Offline Memory Consolidation for Language Agents + +--- +type: paper +title: "Auto-Dreamer: Learning Offline Memory Consolidation for Language Agents" +authors: Chongrui Ye, Yuxiang Liu, Yu Wang, Haofei Yu, Yining Zhao, Ge Liu, Julian McAuley, Jiaxuan You +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.20616 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-20 +updated_at: 2026-05-20 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory, language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, language-agent +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.20616 diff --git a/papers/items/2026-2605-20833-memgym-a-long-horizon-memory-environment-for-llm-agents.md b/papers/items/2026-2605-20833-memgym-a-long-horizon-memory-environment-for-llm-agents.md new file mode 100644 index 0000000..6bbe775 --- /dev/null +++ b/papers/items/2026-2605-20833-memgym-a-long-horizon-memory-environment-for-llm-agents.md @@ -0,0 +1,66 @@ +# Paper: MemGym: a Long-Horizon Memory Environment for LLM Agents + +--- +type: paper +title: "MemGym: a Long-Horizon Memory Environment for LLM Agents" +authors: Wujiang Xu, Yu Wang, Kai Mei, Kaiqu Liang, Zhenting Wang, Mingyu Jin, Han Zhang, Shi-Xiong Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.20833 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-20 +updated_at: 2026-05-20 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - embodied-agent + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 26 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, coding-agent, computer-use, embodied-agent, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 26 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.20833 diff --git a/papers/items/2026-2605-20874-governance-by-construction-for-generalist-agents.md b/papers/items/2026-2605-20874-governance-by-construction-for-generalist-agents.md new file mode 100644 index 0000000..81431bb --- /dev/null +++ b/papers/items/2026-2605-20874-governance-by-construction-for-generalist-agents.md @@ -0,0 +1,64 @@ +# Paper: Governance by Construction for Generalist Agents + +--- +type: paper +title: Governance by Construction for Generalist Agents +authors: Segev Shlomov, Iftach Shoham, Alon Oved, Ido Levy, Sami Marreed, Harold Ship, Offer Akrabi, Sergey Zeltyn, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.20874 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-20 +updated_at: 2026-05-20 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-safety, computer-use, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.20874 diff --git a/papers/items/2026-2605-21240-apex-autonomous-policy-exploration-for-self-evolving-llm-agents.md b/papers/items/2026-2605-21240-apex-autonomous-policy-exploration-for-self-evolving-llm-agents.md new file mode 100644 index 0000000..0cf88b5 --- /dev/null +++ b/papers/items/2026-2605-21240-apex-autonomous-policy-exploration-for-self-evolving-llm-agents.md @@ -0,0 +1,63 @@ +# Paper: APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents + +--- +type: paper +title: "APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents" +authors: Yibo Li, Jiashuo Yang, Zhi Zheng, Zhiyuan Hu, Yuan Sui, Shizun Wang, Yufei He, Bryan Hooi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.21240 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-20 +updated_at: 2026-05-20 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, memory, planning, reasoning, tool-use +- arXiv categories: cs.LG, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.21240 diff --git a/papers/items/2026-2605-21740-smdd-bench-can-llms-solve-real-world-small-molecule-drug-design-tasks.md b/papers/items/2026-2605-21740-smdd-bench-can-llms-solve-real-world-small-molecule-drug-design-tasks.md new file mode 100644 index 0000000..6c686fa --- /dev/null +++ b/papers/items/2026-2605-21740-smdd-bench-can-llms-solve-real-world-small-molecule-drug-design-tasks.md @@ -0,0 +1,62 @@ +# Paper: SMDD-Bench: Can LLMs Solve Real-World Small Molecule Drug Design Tasks? + +--- +type: paper +title: "SMDD-Bench: Can LLMs Solve Real-World Small Molecule Drug Design Tasks?" +authors: Kevin Han, Renfei Zhang, Kathy Wei, Hamed Mahdavi, Niloofar Mireshghallah, Amir Barati Farimani +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.21740 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-20 +updated_at: 2026-05-24 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.21740 diff --git a/papers/items/2026-2605-22154-idlespec-exploiting-idle-time-via-speculative-planning-for-llm-agents.md b/papers/items/2026-2605-22154-idlespec-exploiting-idle-time-via-speculative-planning-for-llm-agents.md new file mode 100644 index 0000000..c880a67 --- /dev/null +++ b/papers/items/2026-2605-22154-idlespec-exploiting-idle-time-via-speculative-planning-for-llm-agents.md @@ -0,0 +1,63 @@ +# Paper: IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents + +--- +type: paper +title: "IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents" +authors: Daewon Choi, Kyunghyun Park, Woomin Song, Saket Dingliwal, Sai Muralidhar Jayanthi, Jinwoo Shin, Aram Galstyan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.22154 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-21 +updated_at: 2026-05-21 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.22154 diff --git a/papers/items/2026-2605-22321-benchmarking-autonomous-agents-against-temporal-spatial-and-semantic-evasions.md b/papers/items/2026-2605-22321-benchmarking-autonomous-agents-against-temporal-spatial-and-semantic-evasions.md new file mode 100644 index 0000000..7093c62 --- /dev/null +++ b/papers/items/2026-2605-22321-benchmarking-autonomous-agents-against-temporal-spatial-and-semantic-evasions.md @@ -0,0 +1,64 @@ +# Paper: Benchmarking Autonomous Agents against Temporal, Spatial, and Semantic Evasions + +--- +type: paper +title: Benchmarking Autonomous Agents against Temporal, Spatial, and Semantic Evasions +authors: Jianan Ma, Xiaohu Du, Ruixiao Lin, Yaoxiang Bian, Jialuo Chen, Jingyi Wang, Xiaofang Yang, Shiwen Cui, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.22321 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-21 +updated_at: 2026-05-21 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use +- arXiv categories: cs.CR, cs.AI, cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.22321 diff --git a/papers/items/2026-2605-22643-boiling-the-frog-a-multi-turn-benchmark-for-agentic-safety.md b/papers/items/2026-2605-22643-boiling-the-frog-a-multi-turn-benchmark-for-agentic-safety.md new file mode 100644 index 0000000..6536f70 --- /dev/null +++ b/papers/items/2026-2605-22643-boiling-the-frog-a-multi-turn-benchmark-for-agentic-safety.md @@ -0,0 +1,63 @@ +# Paper: Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety + +--- +type: paper +title: "Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety" +authors: Piercosma Bisconti, Matteo Prandi, Federico Pierucci, Federico Sartore, Enrico Panai, Laura Caroli, Yue Zhu, Adam Leon Smith, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.22643 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-21 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.22643 diff --git a/papers/items/2026-2605-23067-what-training-data-teaches-rl-memory-agents-an-empirical-study-of-curriculum-eff.md b/papers/items/2026-2605-23067-what-training-data-teaches-rl-memory-agents-an-empirical-study-of-curriculum-eff.md new file mode 100644 index 0000000..f0abc66 --- /dev/null +++ b/papers/items/2026-2605-23067-what-training-data-teaches-rl-memory-agents-an-empirical-study-of-curriculum-eff.md @@ -0,0 +1,60 @@ +# Paper: What Training Data Teaches RL Memory Agents: An Empirical Study of Curriculum Effects in Memory-Augmented QA + +--- +type: paper +title: "What Training Data Teaches RL Memory Agents: An Empirical Study of Curriculum Effects in Memory-Augmented QA" +authors: Xinjie He, Zhiyuan Lin, Su Liu, Jialun Wu, Qiyang Xie, Weikai Zhou, Shuai Xiao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.23067 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-21 +updated_at: 2026-05-21 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, reasoning +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.23067 diff --git a/papers/items/2026-2605-23574-push-your-agent-measuring-and-enforcing-quantitative-goal-persistence-in-long-ho.md b/papers/items/2026-2605-23574-push-your-agent-measuring-and-enforcing-quantitative-goal-persistence-in-long-ho.md new file mode 100644 index 0000000..6e5b25d --- /dev/null +++ b/papers/items/2026-2605-23574-push-your-agent-measuring-and-enforcing-quantitative-goal-persistence-in-long-ho.md @@ -0,0 +1,64 @@ +# Paper: Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents + +--- +type: paper +title: "Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents" +authors: Yuandao Cai, Yuzhang Zhu, Liyou Gao, Wensheng Tang, Shengchao Qin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.23574 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-22 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, coding-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.LG, cs.SE +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.23574 diff --git a/papers/items/2026-2605-23636-rf-instrument-agent-rfia-empowering-rf-instruments-with-natural-language-underst.md b/papers/items/2026-2605-23636-rf-instrument-agent-rfia-empowering-rf-instruments-with-natural-language-underst.md new file mode 100644 index 0000000..64277a6 --- /dev/null +++ b/papers/items/2026-2605-23636-rf-instrument-agent-rfia-empowering-rf-instruments-with-natural-language-underst.md @@ -0,0 +1,63 @@ +# Paper: RF Instrument Agent (RFIA): Empowering RF Instruments with Natural Language Understanding, Scheduling and Execution of Complex Tasks + +--- +type: paper +title: "RF Instrument Agent (RFIA): Empowering RF Instruments with Natural Language Understanding, Scheduling and Execution of Complex Tasks" +authors: Chunhui Li, Wei Fan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.23636 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-22 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, agent-safety, planning, rag, tool-use, workflow-agent +- arXiv categories: eess.SY +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.23636 diff --git a/papers/items/2026-2605-23723-memaudit-post-hoc-auditing-of-poisoned-agent-memory-via-causal-attribution-and-s.md b/papers/items/2026-2605-23723-memaudit-post-hoc-auditing-of-poisoned-agent-memory-via-causal-attribution-and-s.md new file mode 100644 index 0000000..163fcb0 --- /dev/null +++ b/papers/items/2026-2605-23723-memaudit-post-hoc-auditing-of-poisoned-agent-memory-via-causal-attribution-and-s.md @@ -0,0 +1,63 @@ +# Paper: MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection + +--- +type: paper +title: "MemAudit: Post-hoc Auditing of Poisoned Agent Memory via Causal Attribution and Structural Anomaly Detection" +authors: Zhewen Tan, Yilun Yao, Huiyan Jin, Wenhan Yu, Guoan Wang, Mengyuan Fan, liang lu, Feng Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.23723 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-22 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, memory, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.23723 diff --git a/papers/items/2026-2605-23899-from-raw-experience-to-skill-consumption-a-systematic-study-of-model-generated-a.md b/papers/items/2026-2605-23899-from-raw-experience-to-skill-consumption-a-systematic-study-of-model-generated-a.md new file mode 100644 index 0000000..a33eed6 --- /dev/null +++ b/papers/items/2026-2605-23899-from-raw-experience-to-skill-consumption-a-systematic-study-of-model-generated-a.md @@ -0,0 +1,62 @@ +# Paper: From Raw Experience to Skill Consumption: A Systematic Study of Model-Generated Agent Skills + +--- +type: paper +title: "From Raw Experience to Skill Consumption: A Systematic Study of Model-Generated Agent Skills" +authors: Zisu Huang, Jingwen Xu, Yifan Yang, Ziyang Gong, Qihao Yang, Muzhao Tian, Xiaohua Wang, Changze Lv, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.23899 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-22 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, rag, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.23899 diff --git a/papers/items/2026-2605-23986-memforest-an-efficient-agent-memory-system-with-hierarchical-temporal-indexing.md b/papers/items/2026-2605-23986-memforest-an-efficient-agent-memory-system-with-hierarchical-temporal-indexing.md new file mode 100644 index 0000000..63c3914 --- /dev/null +++ b/papers/items/2026-2605-23986-memforest-an-efficient-agent-memory-system-with-hierarchical-temporal-indexing.md @@ -0,0 +1,62 @@ +# Paper: MemForest: An Efficient Agent Memory System with Hierarchical Temporal Indexing + +--- +type: paper +title: "MemForest: An Efficient Agent Memory System with Hierarchical Temporal Indexing" +authors: Han Chen, Zining Zhang, Wenqi Pei, Bingsheng He, Ming Wu, Jason Zeng, Michael Heinrich, Wei Wu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.23986 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-16 +updated_at: 2026-05-16 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DB + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, tool-use +- arXiv categories: cs.DB, cs.AI, cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.23986 diff --git a/papers/items/2026-2605-24069-when-the-manual-lies-a-realistic-benchmark-to-evaluate-mcp-poisoning-attacks-for.md b/papers/items/2026-2605-24069-when-the-manual-lies-a-realistic-benchmark-to-evaluate-mcp-poisoning-attacks-for.md new file mode 100644 index 0000000..42ba980 --- /dev/null +++ b/papers/items/2026-2605-24069-when-the-manual-lies-a-realistic-benchmark-to-evaluate-mcp-poisoning-attacks-for.md @@ -0,0 +1,62 @@ +# Paper: When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents + +--- +type: paper +title: "When the Manual Lies: A Realistic Benchmark to Evaluate MCP Poisoning Attacks for LLM Agents" +authors: Shi Liu, Xuehai Tang, Xikang Yang, Liang Lin, Biyu Zhou, Wenjie Xiao, Wantao Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.24069 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-22 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.24069 diff --git a/papers/items/2026-2605-24216-agent-tom-learning-to-monitor-autonomous-llm-agents-via-theory-of-mind-reasoning.md b/papers/items/2026-2605-24216-agent-tom-learning-to-monitor-autonomous-llm-agents-via-theory-of-mind-reasoning.md new file mode 100644 index 0000000..e51e28a --- /dev/null +++ b/papers/items/2026-2605-24216-agent-tom-learning-to-monitor-autonomous-llm-agents-via-theory-of-mind-reasoning.md @@ -0,0 +1,67 @@ +# Paper: Agent-ToM: Learning to Monitor Autonomous LLM Agents via Theory-of-Mind Reasoning + +--- +type: paper +title: "Agent-ToM: Learning to Monitor Autonomous LLM Agents via Theory-of-Mind Reasoning" +authors: Nesreen K. Ahmed, Nima Nafisi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.24216 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-22 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, planning, reasoning, tool-use +- arXiv categories: cs.LG, cs.AI, cs.CL, cs.CR +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.24216 diff --git a/papers/items/2026-2605-24219-beyond-final-answers-auditing-trajectory-level-hallucinations-in-multi-agent-ind.md b/papers/items/2026-2605-24219-beyond-final-answers-auditing-trajectory-level-hallucinations-in-multi-agent-ind.md new file mode 100644 index 0000000..a349412 --- /dev/null +++ b/papers/items/2026-2605-24219-beyond-final-answers-auditing-trajectory-level-hallucinations-in-multi-agent-ind.md @@ -0,0 +1,61 @@ +# Paper: Beyond Final Answers: Auditing Trajectory-Level Hallucinations in Multi-Agent Industrial Workflows + +--- +type: paper +title: "Beyond Final Answers: Auditing Trajectory-Level Hallucinations in Multi-Agent Industrial Workflows" +authors: Harshada Badave, Santosh Borse, Andrea Gomez, Harshitha Narahari, Sara Carter, Vishwa Bhatt, Aishani Rachakonda, Shuxin Lin, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.24219 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-22 +updated_at: 2026-05-26 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, multi-agent, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.24219 diff --git a/papers/items/2026-2605-24220-polar-agentic-rl-on-any-harness-at-scale.md b/papers/items/2026-2605-24220-polar-agentic-rl-on-any-harness-at-scale.md new file mode 100644 index 0000000..4b29ebc --- /dev/null +++ b/papers/items/2026-2605-24220-polar-agentic-rl-on-any-harness-at-scale.md @@ -0,0 +1,61 @@ +# Paper: Polar: Agentic RL on Any Harness at Scale + +--- +type: paper +title: "Polar: Agentic RL on Any Harness at Scale" +authors: Binfeng Xu, Hao Zhang, Shaokun Zhang, Songyang Han, Mingjie Liu, Jian Hu, Shizhe Diao, Zhenghui Jin, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.24220 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-22 +updated_at: 2026-05-22 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, coding-agent, multi-agent, tool-use +- arXiv categories: cs.DC +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.24220 diff --git a/papers/items/2026-2605-24309-reframing-llm-agent-security-as-an-agent-human-interaction-problem.md b/papers/items/2026-2605-24309-reframing-llm-agent-security-as-an-agent-human-interaction-problem.md new file mode 100644 index 0000000..f4873f7 --- /dev/null +++ b/papers/items/2026-2605-24309-reframing-llm-agent-security-as-an-agent-human-interaction-problem.md @@ -0,0 +1,60 @@ +# Paper: Reframing LLM Agent Security as an Agent-Human Interaction Problem + +--- +type: paper +title: Reframing LLM Agent Security as an Agent-Human Interaction Problem +authors: Peiran Wang, Ying Li, Yuan Tian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.24309 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-23 +updated_at: 2026-05-23 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.24309 diff --git a/papers/items/2026-2605-24659-iterinject-indirect-prompt-injection-against-llm-agents-via-feedback-guided-iter.md b/papers/items/2026-2605-24659-iterinject-indirect-prompt-injection-against-llm-agents-via-feedback-guided-iter.md new file mode 100644 index 0000000..1e7b845 --- /dev/null +++ b/papers/items/2026-2605-24659-iterinject-indirect-prompt-injection-against-llm-agents-via-feedback-guided-iter.md @@ -0,0 +1,62 @@ +# Paper: IterInject: Indirect Prompt Injection Against LLM Agents via Feedback-Guided Iterative Optimization + +--- +type: paper +title: "IterInject: Indirect Prompt Injection Against LLM Agents via Feedback-Guided Iterative Optimization" +authors: Zixuan Chen, Jiaxiang Chen, Li Luo, Ke Xu, Xiaoxiang Huang, Tanfeng Sun, Xinghao Jiang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.24659 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-23 +updated_at: 2026-05-23 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-safety, coding-agent, computer-use, planning, tool-use +- arXiv categories: cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.24659 diff --git a/papers/items/2026-2605-24812-core-code-collaborative-reinforcement-learning-for-code-generation.md b/papers/items/2026-2605-24812-core-code-collaborative-reinforcement-learning-for-code-generation.md new file mode 100644 index 0000000..5466229 --- /dev/null +++ b/papers/items/2026-2605-24812-core-code-collaborative-reinforcement-learning-for-code-generation.md @@ -0,0 +1,64 @@ +# Paper: CoRe-Code: Collaborative Reinforcement Learning for Code Generation + +--- +type: paper +title: "CoRe-Code: Collaborative Reinforcement Learning for Code Generation" +authors: Zhihao Dou, Qinjian Zhao, Zhongwei Wan, Xiaoyu Xia, Sumon Biswas +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.24812 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-24 +updated_at: 2026-05-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - multi-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, multi-agent, planning, rag +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.24812 diff --git a/papers/items/2026-2605-25141-llm-agent-based-renewable-energy-forecasting-using-edge-and-iot-data-a-review-of.md b/papers/items/2026-2605-25141-llm-agent-based-renewable-energy-forecasting-using-edge-and-iot-data-a-review-of.md new file mode 100644 index 0000000..c3a61e3 --- /dev/null +++ b/papers/items/2026-2605-25141-llm-agent-based-renewable-energy-forecasting-using-edge-and-iot-data-a-review-of.md @@ -0,0 +1,65 @@ +# Paper: LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support + +--- +type: paper +title: LLM Agent Based Renewable Energy Forecasting Using Edge and IoT Data A Review of Solar Wind Weather and Grid Aware Decision Support +authors: Pavan Manjunath, Thomas Pruefer +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.25141 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-24 +updated_at: 2026-05-24 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CL, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.25141 diff --git a/papers/items/2026-2605-25200-grouptravelbench-benchmarking-llm-agents-on-multi-person-travel-planning.md b/papers/items/2026-2605-25200-grouptravelbench-benchmarking-llm-agents-on-multi-person-travel-planning.md new file mode 100644 index 0000000..89221a6 --- /dev/null +++ b/papers/items/2026-2605-25200-grouptravelbench-benchmarking-llm-agents-on-multi-person-travel-planning.md @@ -0,0 +1,61 @@ +# Paper: GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning + +--- +type: paper +title: "GroupTravelBench: Benchmarking LLM Agents on Multi-Person Travel Planning" +authors: Xiang Cheng, Yulan Hu, Lulu Zheng, Zheng Pan, Xin Li, Yong Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.25200 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-24 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.25200 diff --git a/papers/items/2026-2605-25310-tool-call-dependency-structure-is-linearly-decodable-in-llm-agent-residual-strea.md b/papers/items/2026-2605-25310-tool-call-dependency-structure-is-linearly-decodable-in-llm-agent-residual-strea.md new file mode 100644 index 0000000..5fde601 --- /dev/null +++ b/papers/items/2026-2605-25310-tool-call-dependency-structure-is-linearly-decodable-in-llm-agent-residual-strea.md @@ -0,0 +1,60 @@ +# Paper: Tool-Call Dependency Structure is Linearly Decodable in LLM Agent Residual Streams + +--- +type: paper +title: Tool-Call Dependency Structure is Linearly Decodable in LLM Agent Residual Streams +authors: Tianda Sun, Dimitar Kazakov +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.25310 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-25 +updated_at: 2026-05-25 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, tool-use +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.25310 diff --git a/papers/items/2026-2605-25393-decision-making-with-lightweight-confidence-aware-language-model-for-autonomous-.md b/papers/items/2026-2605-25393-decision-making-with-lightweight-confidence-aware-language-model-for-autonomous-.md new file mode 100644 index 0000000..2614908 --- /dev/null +++ b/papers/items/2026-2605-25393-decision-making-with-lightweight-confidence-aware-language-model-for-autonomous-.md @@ -0,0 +1,64 @@ +# Paper: Decision-Making with Lightweight Confidence-Aware Language Model for Autonomous Driving + +--- +type: paper +title: Decision-Making with Lightweight Confidence-Aware Language Model for Autonomous Driving +authors: Ruoyu Yao, Ruiguo Zhong, Pei Liu, Mingxing Peng, Rui Yang, Jun Ma +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.25393 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-25 +updated_at: 2026-05-25 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, multi-agent, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.RO +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.25393 diff --git a/papers/items/2026-2605-25435-security-of-openclaw-agents-fundamentals-attacks-and-countermeasures.md b/papers/items/2026-2605-25435-security-of-openclaw-agents-fundamentals-attacks-and-countermeasures.md new file mode 100644 index 0000000..02198f6 --- /dev/null +++ b/papers/items/2026-2605-25435-security-of-openclaw-agents-fundamentals-attacks-and-countermeasures.md @@ -0,0 +1,63 @@ +# Paper: Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures + +--- +type: paper +title: "Security of OpenClaw Agents: Fundamentals, Attacks, and Countermeasures" +authors: Yuntao Wang, Jianle Ba, Han Liu, Yanghe Pan, Jintao Wei, Zhou Su, Tom H. Luan, Linkang Du +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.25435 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-25 +updated_at: 2026-05-25 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - memory + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-safety, computer-use, memory, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.25435 diff --git a/papers/items/2026-2605-25920-can-llms-time-travel-enhancing-temporal-consistency-in-legal-agentic-search-thro.md b/papers/items/2026-2605-25920-can-llms-time-travel-enhancing-temporal-consistency-in-legal-agentic-search-thro.md new file mode 100644 index 0000000..205fac8 --- /dev/null +++ b/papers/items/2026-2605-25920-can-llms-time-travel-enhancing-temporal-consistency-in-legal-agentic-search-thro.md @@ -0,0 +1,61 @@ +# Paper: Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning + +--- +type: paper +title: Can LLMs Time Travel? Enhancing Temporal Consistency in Legal Agentic Search through Reinforcement Learning +authors: Wei Fan, Yining Zhou, Mufan Zhang, Yanbing Weng, Yiran HU, Tianshi Zheng, Baixuan Xu, Chunyang Li, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.25920 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-25 +updated_at: 2026-05-25 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, rag, reasoning +- arXiv categories: cs.CL, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.25920 diff --git a/papers/items/2026-2605-26165-tool-schema-compression-enables-agentic-rag-under-constrained-context-budgets.md b/papers/items/2026-2605-26165-tool-schema-compression-enables-agentic-rag-under-constrained-context-budgets.md new file mode 100644 index 0000000..834c68d --- /dev/null +++ b/papers/items/2026-2605-26165-tool-schema-compression-enables-agentic-rag-under-constrained-context-budgets.md @@ -0,0 +1,62 @@ +# Paper: Tool-Schema Compression Enables Agentic RAG Under Constrained Context Budgets + +--- +type: paper +title: Tool-Schema Compression Enables Agentic RAG Under Constrained Context Budgets +authors: Furkan Sakizli +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.26165 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-24 +updated_at: 2026-05-24 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, rag, tool-use +- arXiv categories: cs.SE, cs.AI, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.26165 diff --git a/papers/items/2026-2605-26252-is-agent-memory-a-database-rethinking-data-foundations-for-long-term-ai-agent-me.md b/papers/items/2026-2605-26252-is-agent-memory-a-database-rethinking-data-foundations-for-long-term-ai-agent-me.md new file mode 100644 index 0000000..c642aa2 --- /dev/null +++ b/papers/items/2026-2605-26252-is-agent-memory-a-database-rethinking-data-foundations-for-long-term-ai-agent-me.md @@ -0,0 +1,62 @@ +# Paper: Is Agent Memory a Database? Rethinking Data Foundations for Long-Term AI Agent Memory + +--- +type: paper +title: Is Agent Memory a Database? Rethinking Data Foundations for Long-Term AI Agent Memory +authors: Abdelghny Orogat, Essam Mansour +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.26252 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-25 +updated_at: 2026-05-25 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.DB +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.AI, cs.DB +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.26252 diff --git a/papers/items/2026-2605-26305-experiments-in-agentic-ai-for-science.md b/papers/items/2026-2605-26305-experiments-in-agentic-ai-for-science.md new file mode 100644 index 0000000..54dece0 --- /dev/null +++ b/papers/items/2026-2605-26305-experiments-in-agentic-ai-for-science.md @@ -0,0 +1,63 @@ +# Paper: Experiments in Agentic AI for Science + +--- +type: paper +title: Experiments in Agentic AI for Science +authors: Judy Fox, Geoffrey Fox +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.26305 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-25 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - eess.SY + - hep-ph +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, rag-agent +- inferred topics: rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, eess.SY, hep-ph +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.26305 diff --git a/papers/items/2026-2605-26497-aligning-provenance-with-authorization-a-dual-graph-defense-for-llm-agents.md b/papers/items/2026-2605-26497-aligning-provenance-with-authorization-a-dual-graph-defense-for-llm-agents.md new file mode 100644 index 0000000..39c99bb --- /dev/null +++ b/papers/items/2026-2605-26497-aligning-provenance-with-authorization-a-dual-graph-defense-for-llm-agents.md @@ -0,0 +1,60 @@ +# Paper: Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents + +--- +type: paper +title: "Aligning Provenance with Authorization: A Dual-Graph Defense for LLM Agents" +authors: Peiran Wang, Ying Li, Yuan Tian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.26497 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-05-26 +status: queued +relevance: high +topics: + - agent-safety + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, reasoning, tool-use +- arXiv categories: cs.CR +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.26497 diff --git a/papers/items/2026-2605-26720-towards-feedback-to-plan-decisions-for-self-evolving-llm-agents-in-cuda-kernel-g.md b/papers/items/2026-2605-26720-towards-feedback-to-plan-decisions-for-self-evolving-llm-agents-in-cuda-kernel-g.md new file mode 100644 index 0000000..60b1ddd --- /dev/null +++ b/papers/items/2026-2605-26720-towards-feedback-to-plan-decisions-for-self-evolving-llm-agents-in-cuda-kernel-g.md @@ -0,0 +1,61 @@ +# Paper: Towards Feedback-to-Plan Decisions for Self-Evolving LLM Agents in CUDA Kernel Generation + +--- +type: paper +title: Towards Feedback-to-Plan Decisions for Self-Evolving LLM Agents in CUDA Kernel Generation +authors: Yee Hin Chong, Jiaming Wu, Youhui Zhang, Peng Qu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.26720 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-05-26 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.26720 diff --git a/papers/items/2026-2605-26926-from-norms-to-indicators-n2i-rag-an-agentic-retrieval-augmented-generation-frame.md b/papers/items/2026-2605-26926-from-norms-to-indicators-n2i-rag-an-agentic-retrieval-augmented-generation-frame.md new file mode 100644 index 0000000..416c24a --- /dev/null +++ b/papers/items/2026-2605-26926-from-norms-to-indicators-n2i-rag-an-agentic-retrieval-augmented-generation-frame.md @@ -0,0 +1,61 @@ +# Paper: From Norms to Indicators (N2I-RAG): An Agentic Retrieval-Augmented Generation Framework for Legal Indicator Computation + +--- +type: paper +title: "From Norms to Indicators (N2I-RAG): An Agentic Retrieval-Augmented Generation Framework for Legal Indicator Computation" +authors: Youssef Al Mouatamid, Marie Bonnin, Jihad Zahir +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.26926 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-05-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, agent-safety, planning, rag +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.26926 diff --git a/papers/items/2026-2605-27123-rethinking-agentic-rag-toward-llm-driven-logical-retrieval-beyond-embeddings.md b/papers/items/2026-2605-27123-rethinking-agentic-rag-toward-llm-driven-logical-retrieval-beyond-embeddings.md new file mode 100644 index 0000000..c9b69cd --- /dev/null +++ b/papers/items/2026-2605-27123-rethinking-agentic-rag-toward-llm-driven-logical-retrieval-beyond-embeddings.md @@ -0,0 +1,60 @@ +# Paper: Rethinking Agentic RAG: Toward LLM-Driven Logical Retrieval Beyond Embeddings + +--- +type: paper +title: "Rethinking Agentic RAG: Toward LLM-Driven Logical Retrieval Beyond Embeddings" +authors: Yuqi Zeng, Qixiang Deng, Yulei Wan, Ruiquan Jiang, Xiaoqing Zheng, Xuanjing Huang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.27123 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-05-26 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, computer-use, rag +- arXiv categories: cs.IR +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.27123 diff --git a/papers/items/2026-2605-27134-scaling-benchmarking-and-reasoning-of-vision-language-agents-for-mobile-gui-navi.md b/papers/items/2026-2605-27134-scaling-benchmarking-and-reasoning-of-vision-language-agents-for-mobile-gui-navi.md new file mode 100644 index 0000000..1674411 --- /dev/null +++ b/papers/items/2026-2605-27134-scaling-benchmarking-and-reasoning-of-vision-language-agents-for-mobile-gui-navi.md @@ -0,0 +1,63 @@ +# Paper: Scaling, Benchmarking, and Reasoning of Vision-Language Agents for Mobile GUI Navigation + +--- +type: paper +title: Scaling, Benchmarking, and Reasoning of Vision-Language Agents for Mobile GUI Navigation +authors: Heng Qu, Yike Liu, Renren Jin, Wenzong Zhang, Pengzhi Gao, Wei Liu, Jian Luan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.27134 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-05-26 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, computer-use, embodied-agent, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.27134 diff --git a/papers/items/2026-2605-27240-enpmr-bench-benchmarking-proactive-memory-retrieval-for-emotional-support-agents.md b/papers/items/2026-2605-27240-enpmr-bench-benchmarking-proactive-memory-retrieval-for-emotional-support-agents.md new file mode 100644 index 0000000..2a02d49 --- /dev/null +++ b/papers/items/2026-2605-27240-enpmr-bench-benchmarking-proactive-memory-retrieval-for-emotional-support-agents.md @@ -0,0 +1,62 @@ +# Paper: ENPMR-Bench: Benchmarking Proactive Memory Retrieval for Emotional Support Agents + +--- +type: paper +title: "ENPMR-Bench: Benchmarking Proactive Memory Retrieval for Emotional Support Agents" +authors: Xing Fu, Yulin Hu, Mengtong Ji, Haozhen Li, Yixin Sun, Weixiang Zhao, Yanyan Zhao, Bing Qin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.27240 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-05-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use +- arXiv categories: cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.27240 diff --git a/papers/items/2026-2605-27333-finharness-an-inline-lifecycle-safety-harness-for-finance-llm-agents.md b/papers/items/2026-2605-27333-finharness-an-inline-lifecycle-safety-harness-for-finance-llm-agents.md new file mode 100644 index 0000000..30f9464 --- /dev/null +++ b/papers/items/2026-2605-27333-finharness-an-inline-lifecycle-safety-harness-for-finance-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: FinHarness: An Inline Lifecycle Safety Harness for Finance LLM Agents + +--- +type: paper +title: "FinHarness: An Inline Lifecycle Safety Harness for Finance LLM Agents" +authors: Haoxuan Jia, Yang Liu, Bin Chong, Yingguang Yang, Yancheng Chen, Jiayu Liang, Qian Li, Hanning Lu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.27333 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-05-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.27333 diff --git a/papers/items/2026-2605-27366-muse-autoskill-self-evolving-agents-via-skill-creation-memory-management-and-eva.md b/papers/items/2026-2605-27366-muse-autoskill-self-evolving-agents-via-skill-creation-memory-management-and-eva.md new file mode 100644 index 0000000..624d49e --- /dev/null +++ b/papers/items/2026-2605-27366-muse-autoskill-self-evolving-agents-via-skill-creation-memory-management-and-eva.md @@ -0,0 +1,62 @@ +# Paper: MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation + +--- +type: paper +title: "MUSE-Autoskill: Self-Evolving Agents via Skill Creation, Memory, Management, and Evaluation" +authors: Huawei Lin, Peng Li, Jie Song, Fuxin Jiang, Tieying Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.27366 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory +- arXiv categories: cs.AI, cs.CL, cs.LG, cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.27366 diff --git a/papers/items/2026-2605-27690-traces-proactive-safety-auditing-for-multi-turn-llm-agents-via-trajectory-state-.md b/papers/items/2026-2605-27690-traces-proactive-safety-auditing-for-multi-turn-llm-agents-via-trajectory-state-.md new file mode 100644 index 0000000..cf14f11 --- /dev/null +++ b/papers/items/2026-2605-27690-traces-proactive-safety-auditing-for-multi-turn-llm-agents-via-trajectory-state-.md @@ -0,0 +1,63 @@ +# Paper: TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling + +--- +type: paper +title: "TRACES: Proactive Safety Auditing for Multi-Turn LLM Agents via Trajectory-State Modeling" +authors: Jiaqian Li, Yanshu Li, Boxuan Zhang, Ruixiang Tang, Kuan-Hao Huang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.27690 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-05-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, planning, tool-use +- arXiv categories: cs.CL, cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.27690 diff --git a/papers/items/2026-2605-27762-peam-parametric-embodied-agent-memory-through-contrastive-internalization-of-exp.md b/papers/items/2026-2605-27762-peam-parametric-embodied-agent-memory-through-contrastive-internalization-of-exp.md new file mode 100644 index 0000000..d556259 --- /dev/null +++ b/papers/items/2026-2605-27762-peam-parametric-embodied-agent-memory-through-contrastive-internalization-of-exp.md @@ -0,0 +1,64 @@ +# Paper: PEAM: Parametric Embodied Agent Memory through Contrastive Internalization of Experience in Minecraft + +--- +type: paper +title: "PEAM: Parametric Embodied Agent Memory through Contrastive Internalization of Experience in Minecraft" +authors: Yuchen Guo, Junli Gong, Weicheng Wang, Hongmin Cai, Yiu-ming Cheung, Weifeng Su +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.27762 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-26 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, embodied-agent, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.27762 diff --git a/papers/items/2026-2605-27825-mrmmia-membership-inference-attacks-on-memory-in-chat-agents.md b/papers/items/2026-2605-27825-mrmmia-membership-inference-attacks-on-memory-in-chat-agents.md new file mode 100644 index 0000000..8e274a3 --- /dev/null +++ b/papers/items/2026-2605-27825-mrmmia-membership-inference-attacks-on-memory-in-chat-agents.md @@ -0,0 +1,63 @@ +# Paper: MRMMIA: Membership Inference Attacks on Memory in Chat Agents + +--- +type: paper +title: "MRMMIA: Membership Inference Attacks on Memory in Chat Agents" +authors: Kai Chen, Yan Pang, Tianhao Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.27825 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-27 +updated_at: 2026-05-27 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use +- arXiv categories: cs.CR, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.27825 diff --git a/papers/items/2026-2605-27935-do-agents-think-deeper-a-mechanistic-investigation-of-layer-wise-dynamics-in-seq.md b/papers/items/2026-2605-27935-do-agents-think-deeper-a-mechanistic-investigation-of-layer-wise-dynamics-in-seq.md new file mode 100644 index 0000000..2912a9c --- /dev/null +++ b/papers/items/2026-2605-27935-do-agents-think-deeper-a-mechanistic-investigation-of-layer-wise-dynamics-in-seq.md @@ -0,0 +1,60 @@ +# Paper: Do Agents Think Deeper? A Mechanistic Investigation of Layer-Wise Dynamics in Sequential Planning + +--- +type: paper +title: Do Agents Think Deeper? A Mechanistic Investigation of Layer-Wise Dynamics in Sequential Planning +authors: Zhenyu Cui, Xiangzhong Luo +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.27935 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-27 +updated_at: 2026-05-27 +status: queued +relevance: high +topics: + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, planning-agent +- inferred topics: planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.27935 diff --git a/papers/items/2026-2605-28046-memcog-from-memory-as-tool-to-memory-as-cognition-in-conversational-agents.md b/papers/items/2026-2605-28046-memcog-from-memory-as-tool-to-memory-as-cognition-in-conversational-agents.md new file mode 100644 index 0000000..06275ea --- /dev/null +++ b/papers/items/2026-2605-28046-memcog-from-memory-as-tool-to-memory-as-cognition-in-conversational-agents.md @@ -0,0 +1,64 @@ +# Paper: MemCog: From Memory-as-Tool to Memory-as-Cognition in Conversational Agents + +--- +type: paper +title: "MemCog: From Memory-as-Tool to Memory-as-Cognition in Conversational Agents" +authors: Zihan Li, Xingyu Fan, Feifei Li, Wenhui Que +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.28046 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-27 +updated_at: 2026-05-27 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, embodied-agent, memory, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.28046 diff --git a/papers/items/2026-2605-28120-legalgraphrag-multi-agent-graph-retrieval-augmented-generation-for-reliable-lega.md b/papers/items/2026-2605-28120-legalgraphrag-multi-agent-graph-retrieval-augmented-generation-for-reliable-lega.md new file mode 100644 index 0000000..cacea7a --- /dev/null +++ b/papers/items/2026-2605-28120-legalgraphrag-multi-agent-graph-retrieval-augmented-generation-for-reliable-lega.md @@ -0,0 +1,64 @@ +# Paper: LegalGraphRAG: Multi-Agent Graph Retrieval-Augmented Generation for Reliable Legal Reasoning + +--- +type: paper +title: "LegalGraphRAG: Multi-Agent Graph Retrieval-Augmented Generation for Reliable Legal Reasoning" +authors: Zerui Chen, Qinggang Zhang, Zhishang Xiang, Zhimin Wei, Linfeng Gao, Xiao Huang, Zhihong Zhang, Jinsong Su +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.28120 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-27 +updated_at: 2026-05-27 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.CL, cs.AI, cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.28120 diff --git a/papers/items/2026-2605-28175-mixture-of-experts-knowledge-graph-retrieval-augmented-generation-for-multi-agen.md b/papers/items/2026-2605-28175-mixture-of-experts-knowledge-graph-retrieval-augmented-generation-for-multi-agen.md new file mode 100644 index 0000000..d6b6f58 --- /dev/null +++ b/papers/items/2026-2605-28175-mixture-of-experts-knowledge-graph-retrieval-augmented-generation-for-multi-agen.md @@ -0,0 +1,61 @@ +# Paper: Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation + +--- +type: paper +title: Mixture-of-Experts Knowledge Graph Retrieval-Augmented Generation for Multi-Agent LLM-based Recommendation +authors: Shijie Wang, Chengyi Liu, Yujuan Ding, Shanru Lin, See-Kiong Ng, Xu Xin, Wenqi Fan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.28175 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-27 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag +- arXiv categories: cs.IR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.28175 diff --git a/papers/items/2026-2605-28424-skill0-5-joint-skill-internalization-and-utilization-for-out-of-distribution-gen.md b/papers/items/2026-2605-28424-skill0-5-joint-skill-internalization-and-utilization-for-out-of-distribution-gen.md new file mode 100644 index 0000000..88ee003 --- /dev/null +++ b/papers/items/2026-2605-28424-skill0-5-joint-skill-internalization-and-utilization-for-out-of-distribution-gen.md @@ -0,0 +1,59 @@ +# Paper: Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning + +--- +type: paper +title: "Skill0.5: Joint Skill Internalization and Utilization for Out-of-Distribution Generalization in Agentic Reinforcement Learning" +authors: Jiapeng Zhu, Jianxiang Yu, Yibo Zhao, Chengcheng Han, Qi Gu, Xunliang Cai, Xiang Li, Weining Qian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.28424 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-27 +updated_at: 2026-05-27 +status: queued +relevance: high +topics: + - agent-safety + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-safety, memory +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.28424 diff --git a/papers/items/2026-2605-28607-adaptive-multimodal-agents-based-framework-for-automatic-workflow-execution.md b/papers/items/2026-2605-28607-adaptive-multimodal-agents-based-framework-for-automatic-workflow-execution.md new file mode 100644 index 0000000..b5f859f --- /dev/null +++ b/papers/items/2026-2605-28607-adaptive-multimodal-agents-based-framework-for-automatic-workflow-execution.md @@ -0,0 +1,65 @@ +# Paper: Adaptive Multimodal Agents-Based Framework for Automatic Workflow Execution + +--- +type: paper +title: Adaptive Multimodal Agents-Based Framework for Automatic Workflow Execution +authors: Susanna Cifani, Mario Luca Bernardi, Marta Cimitile +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.28607 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-27 +updated_at: 2026-05-27 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - multi-agent + - planning + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, computer-use, embodied-agent, multi-agent, planning, rag, workflow-agent +- arXiv categories: cs.AI, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.28607 diff --git a/papers/items/2026-2605-28617-lacuna-safe-agents-as-recursive-program-holes.md b/papers/items/2026-2605-28617-lacuna-safe-agents-as-recursive-program-holes.md new file mode 100644 index 0000000..3a95878 --- /dev/null +++ b/papers/items/2026-2605-28617-lacuna-safe-agents-as-recursive-program-holes.md @@ -0,0 +1,63 @@ +# Paper: LACUNA: Safe Agents as Recursive Program Holes + +--- +type: paper +title: "LACUNA: Safe Agents as Recursive Program Holes" +authors: Yaoyu Zhao, Yichen Xu, Oliver Bračevac, Cao Nguyen Pham, Frank Zhengqing Wu, Martin Odersky +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.28617 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-27 +updated_at: 2026-05-27 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.PL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, rag, tool-use +- arXiv categories: cs.AI, cs.PL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.28617 diff --git a/papers/items/2026-2605-28787-do-agents-need-semantic-metadata-a-comparative-study-in-agentic-data-retrieval.md b/papers/items/2026-2605-28787-do-agents-need-semantic-metadata-a-comparative-study-in-agentic-data-retrieval.md new file mode 100644 index 0000000..ef06505 --- /dev/null +++ b/papers/items/2026-2605-28787-do-agents-need-semantic-metadata-a-comparative-study-in-agentic-data-retrieval.md @@ -0,0 +1,62 @@ +# Paper: Do Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval + +--- +type: paper +title: Do Agents Need Semantic Metadata? A Comparative Study in Agentic Data Retrieval +authors: Shiyu Chen, Tarfah Alrashed, Alon Halevy, Natasha Noy +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.28787 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-27 +updated_at: 2026-05-27 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, rag, tool-use, workflow-agent +- arXiv categories: cs.IR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.28787 diff --git a/papers/items/2026-2605-28835-genesisfunc-multi-agent-data-generation-for-accurate-and-generalizable-function-.md b/papers/items/2026-2605-28835-genesisfunc-multi-agent-data-generation-for-accurate-and-generalizable-function-.md new file mode 100644 index 0000000..bc60b7a --- /dev/null +++ b/papers/items/2026-2605-28835-genesisfunc-multi-agent-data-generation-for-accurate-and-generalizable-function-.md @@ -0,0 +1,62 @@ +# Paper: GenesisFunc: Multi-Agent Data Generation for Accurate and Generalizable Function-Calling + +--- +type: paper +title: "GenesisFunc: Multi-Agent Data Generation for Accurate and Generalizable Function-Calling" +authors: Hao-Xiang Xu, Chong Deng, Jiaqing Liu, Wen Wang, Qian Chen, Lujia Bao, Xiangang Li, Zhen-Hua Ling +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.28835 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-10 +updated_at: 2026-04-10 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, multi-agent, rag, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.28835 diff --git a/papers/items/2026-2605-28850-representation-signatures-and-risk-feedback-alignment-in-llm-trading-agents.md b/papers/items/2026-2605-28850-representation-signatures-and-risk-feedback-alignment-in-llm-trading-agents.md new file mode 100644 index 0000000..c6e1cd4 --- /dev/null +++ b/papers/items/2026-2605-28850-representation-signatures-and-risk-feedback-alignment-in-llm-trading-agents.md @@ -0,0 +1,65 @@ +# Paper: Representation Signatures and Risk-Feedback Alignment in LLM Trading Agents + +--- +type: paper +title: Representation Signatures and Risk-Feedback Alignment in LLM Trading Agents +authors: Weicheng Xue +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.28850 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-16 +updated_at: 2026-05-30 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - memory + - planning + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - q-fin.CP +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-safety, coding-agent, memory, planning, reasoning, tool-use, world-model +- arXiv categories: cs.LG, q-fin.CP +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.28850 diff --git a/papers/items/2026-2605-29341-worldmemarena-evaluating-multimodal-agent-memory-through-action-world-interactio.md b/papers/items/2026-2605-29341-worldmemarena-evaluating-multimodal-agent-memory-through-action-world-interactio.md new file mode 100644 index 0000000..1db343e --- /dev/null +++ b/papers/items/2026-2605-29341-worldmemarena-evaluating-multimodal-agent-memory-through-action-world-interactio.md @@ -0,0 +1,63 @@ +# Paper: WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction + +--- +type: paper +title: "WorldMemArena: Evaluating Multimodal Agent Memory Through Action-World Interaction" +authors: Chengzhi Liu, Yuzhe Yang, Sophia Xiao Pu, Yepeng Liu, Lin Long, Yichen Guo, Nuo Chen, Zhaotian Weng, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.29341 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, rag-agent +- inferred topics: agent-evaluation, memory, planning, rag, tool-use +- arXiv categories: cs.CV, cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.29341 diff --git a/papers/items/2026-2605-29630-entity-collision-a-stratified-protocol-for-attributing-retrieval-lift-in-agent-m.md b/papers/items/2026-2605-29630-entity-collision-a-stratified-protocol-for-attributing-retrieval-lift-in-agent-m.md new file mode 100644 index 0000000..1f8cd06 --- /dev/null +++ b/papers/items/2026-2605-29630-entity-collision-a-stratified-protocol-for-attributing-retrieval-lift-in-agent-m.md @@ -0,0 +1,63 @@ +# Paper: Entity-Collision: A Stratified Protocol for Attributing Retrieval Lift in Agent Memory + +--- +type: paper +title: "Entity-Collision: A Stratified Protocol for Attributing Retrieval Lift in Agent Memory" +authors: Youwang Deng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.29630 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.CL, cs.AI, cs.IR +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.29630 diff --git a/papers/items/2026-2605-29640-vikingmem-a-memory-base-management-system-for-stateful-llm-based-applications.md b/papers/items/2026-2605-29640-vikingmem-a-memory-base-management-system-for-stateful-llm-based-applications.md new file mode 100644 index 0000000..5dbe583 --- /dev/null +++ b/papers/items/2026-2605-29640-vikingmem-a-memory-base-management-system-for-stateful-llm-based-applications.md @@ -0,0 +1,61 @@ +# Paper: VikingMem: A Memory Base Management System for Stateful LLM-based Applications + +--- +type: paper +title: "VikingMem: A Memory Base Management System for Stateful LLM-based Applications" +authors: Jiajie Fu, Junwen Chen, Mengzhao Wang, Aoxiang He, Maojia Sheng, Xiangyu Ke, Yifan Zhu, Yunjun Gao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.29640 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.29640 diff --git a/papers/items/2026-2605-29653-ptcg-bench-can-llm-agents-master-pok-mon-trading-card-game.md b/papers/items/2026-2605-29653-ptcg-bench-can-llm-agents-master-pok-mon-trading-card-game.md new file mode 100644 index 0000000..7c2aa3f --- /dev/null +++ b/papers/items/2026-2605-29653-ptcg-bench-can-llm-agents-master-pok-mon-trading-card-game.md @@ -0,0 +1,58 @@ +# Paper: PTCG-Bench: Can LLM Agents Master Pokémon Trading Card Game? + +--- +type: paper +title: "PTCG-Bench: Can LLM Agents Master Pokémon Trading Card Game?" +authors: Dongdong Hua, Yifei Sun, Renhong Huang, Feng Gao, Chunping Wang, Yang Yang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.29653 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation, autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, autonomous-agent-llm +- inferred topics: agent-evaluation +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.29653 diff --git a/papers/items/2026-2605-29676-notation-matters-a-benchmark-study-of-token-optimized-formats-in-agentic-ai-syst.md b/papers/items/2026-2605-29676-notation-matters-a-benchmark-study-of-token-optimized-formats-in-agentic-ai-syst.md new file mode 100644 index 0000000..b022476 --- /dev/null +++ b/papers/items/2026-2605-29676-notation-matters-a-benchmark-study-of-token-optimized-formats-in-agentic-ai-syst.md @@ -0,0 +1,60 @@ +# Paper: Notation Matters: A Benchmark Study of Token-Optimized Formats in Agentic AI Systems + +--- +type: paper +title: "Notation Matters: A Benchmark Study of Token-Optimized Formats in Agentic AI Systems" +authors: Lorenz Kutschka, Bernhard Geiger +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.29676 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.29676 diff --git a/papers/items/2026-2605-29790-evolve-as-a-team-collaborative-self-evolution-for-llm-based-multi-agent-systems.md b/papers/items/2026-2605-29790-evolve-as-a-team-collaborative-self-evolution-for-llm-based-multi-agent-systems.md new file mode 100644 index 0000000..cf4dac7 --- /dev/null +++ b/papers/items/2026-2605-29790-evolve-as-a-team-collaborative-self-evolution-for-llm-based-multi-agent-systems.md @@ -0,0 +1,61 @@ +# Paper: Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems + +--- +type: paper +title: "Evolve as a Team: Collaborative Self-Evolution for LLM-based Multi-Agent Systems" +authors: Zhezheng Hao, Tianfu Wang, Huanshuo Dong, Ziyan Liu, Hong Wang, Xiankun Lin, Qiang Lin, Can Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.29790 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, multi-agent, planning +- arXiv categories: cs.MA, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.29790 diff --git a/papers/items/2026-2605-29801-agentdog-1-5-a-lightweight-and-scalable-alignment-framework-for-ai-agent-safety-.md b/papers/items/2026-2605-29801-agentdog-1-5-a-lightweight-and-scalable-alignment-framework-for-ai-agent-safety-.md new file mode 100644 index 0000000..b9a63a5 --- /dev/null +++ b/papers/items/2026-2605-29801-agentdog-1-5-a-lightweight-and-scalable-alignment-framework-for-ai-agent-safety-.md @@ -0,0 +1,63 @@ +# Paper: AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security + +--- +type: paper +title: "AgentDoG 1.5: A Lightweight and Scalable Alignment Framework for AI Agent Safety and Security" +authors: Dongrui Liu, Yu Li, Zhonghao Yang, Peng Wang, Guanxu Chen, Yuejin Xie, Qinghua Mao, Wanying Qu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.29801 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-safety + - computer-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.CR + - cs.CV + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-safety, computer-use +- arXiv categories: cs.AI, cs.CL, cs.CR, cs.CV, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.29801 diff --git a/papers/items/2026-2605-29861-towards-verifiable-multimodal-deep-research-a-multi-agent-harness-for-interleave.md b/papers/items/2026-2605-29861-towards-verifiable-multimodal-deep-research-a-multi-agent-harness-for-interleave.md new file mode 100644 index 0000000..cc7b111 --- /dev/null +++ b/papers/items/2026-2605-29861-towards-verifiable-multimodal-deep-research-a-multi-agent-harness-for-interleave.md @@ -0,0 +1,66 @@ +# Paper: Towards Verifiable Multimodal Deep Research: A Multi-Agent Harness for Interleaved Report Generation + +--- +type: paper +title: "Towards Verifiable Multimodal Deep Research: A Multi-Agent Harness for Interleaved Report Generation" +authors: Chenghao Zhang, Guanting Dong, Yufan Liu, Tong Zhao, Xiaoxi Li, Zhicheng Dou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.29861 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, memory, multi-agent, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CL, cs.AI +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.29861 diff --git a/papers/items/2026-2605-29960-hijacking-agent-memory-stealthy-trojan-attacks-through-conversational-interactio.md b/papers/items/2026-2605-29960-hijacking-agent-memory-stealthy-trojan-attacks-through-conversational-interactio.md new file mode 100644 index 0000000..2619939 --- /dev/null +++ b/papers/items/2026-2605-29960-hijacking-agent-memory-stealthy-trojan-attacks-through-conversational-interactio.md @@ -0,0 +1,62 @@ +# Paper: Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction + +--- +type: paper +title: "Hijacking Agent Memory: Stealthy Trojan Attacks Through Conversational Interaction" +authors: Hongtao Wang, Se Yang, Yu Chen, Puzhuo Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.29960 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.29960 diff --git a/papers/items/2026-2605-30058-heart-bench-do-llm-agents-exhibit-human-like-psychology.md b/papers/items/2026-2605-30058-heart-bench-do-llm-agents-exhibit-human-like-psychology.md new file mode 100644 index 0000000..b313460 --- /dev/null +++ b/papers/items/2026-2605-30058-heart-bench-do-llm-agents-exhibit-human-like-psychology.md @@ -0,0 +1,63 @@ +# Paper: HEART-Bench: Do LLM Agents Exhibit Human-like Psychology? + +--- +type: paper +title: "HEART-Bench: Do LLM Agents Exhibit Human-like Psychology?" +authors: Weihan Peng, Chenxu Zhang, Qianao Wang, Yuling Shi, Heng Lian, Qihong Mao, Jiahao Pang, Chunliang Feng, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30058 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30058 diff --git a/papers/items/2026-2605-30090-directorbench-diagnosing-long-form-video-generation-with-personalized-multi-agen.md b/papers/items/2026-2605-30090-directorbench-diagnosing-long-form-video-generation-with-personalized-multi-agen.md new file mode 100644 index 0000000..b74e36b --- /dev/null +++ b/papers/items/2026-2605-30090-directorbench-diagnosing-long-form-video-generation-with-personalized-multi-agen.md @@ -0,0 +1,64 @@ +# Paper: DirectorBench: Diagnosing Long-Form Video Generation with Personalized Multi-Agent Evaluation + +--- +type: paper +title: "DirectorBench: Diagnosing Long-Form Video Generation with Personalized Multi-Agent Evaluation" +authors: Jiamin Chen, Qianben Chen, Jiawen Zhang, Yidi Wu, Yuchen Li, Xiaokun Zhang, Wangchunshu Zhou, Chen Ma +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30090 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, tool-use, workflow-agent +- arXiv categories: cs.CL, cs.CV +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30090 diff --git a/papers/items/2026-2605-30407-exploring-autonomous-agentic-data-engineering-for-model-specialization.md b/papers/items/2026-2605-30407-exploring-autonomous-agentic-data-engineering-for-model-specialization.md new file mode 100644 index 0000000..398fdf6 --- /dev/null +++ b/papers/items/2026-2605-30407-exploring-autonomous-agentic-data-engineering-for-model-specialization.md @@ -0,0 +1,64 @@ +# Paper: Exploring Autonomous Agentic Data Engineering for Model Specialization + +--- +type: paper +title: Exploring Autonomous Agentic Data Engineering for Model Specialization +authors: Yujie Luo, Xiangyuan Ru, Jingsheng Zheng, Jingjing Wang, Yuqi Zhu, Jintian Zhang, Runnan Fang, Kewei Xu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30407 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.IR + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, computer-use, planning, workflow-agent +- arXiv categories: cs.CL, cs.AI, cs.IR, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30407 diff --git a/papers/items/2026-2605-30604-an-organization-scoped-llm-agent-runtime-architecture-for-regulated-cybersecurit.md b/papers/items/2026-2605-30604-an-organization-scoped-llm-agent-runtime-architecture-for-regulated-cybersecurit.md new file mode 100644 index 0000000..b52a5c3 --- /dev/null +++ b/papers/items/2026-2605-30604-an-organization-scoped-llm-agent-runtime-architecture-for-regulated-cybersecurit.md @@ -0,0 +1,67 @@ +# Paper: An Organization-Scoped LLM Agent Runtime Architecture for Regulated Cybersecurity Operations + +--- +type: paper +title: An Organization-Scoped LLM Agent Runtime Architecture for Regulated Cybersecurity Operations +authors: George Fatouros, Georgios Makridis, George Kousiouris, John Soldatos, Dimosthenis Kyriazis +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30604 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.CL + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, memory, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI, cs.CL, cs.IR +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30604 diff --git a/papers/items/2026-2605-30690-elasticmem-latent-memory-as-a-learnable-resource-for-llm-agents.md b/papers/items/2026-2605-30690-elasticmem-latent-memory-as-a-learnable-resource-for-llm-agents.md new file mode 100644 index 0000000..f3c3923 --- /dev/null +++ b/papers/items/2026-2605-30690-elasticmem-latent-memory-as-a-learnable-resource-for-llm-agents.md @@ -0,0 +1,63 @@ +# Paper: ElasticMem: Latent Memory as a Learnable Resource for LLM Agents + +--- +type: paper +title: "ElasticMem: Latent Memory as a Learnable Resource for LLM Agents" +authors: Tao Feng, Chongrui Ye, Tianyang Luo, Jingjun Xu, Xueqiang Xu, Haozhen Zhang, Ge Liu, Jiaxuan You +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30690 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, embodied-agent, memory, planning, rag, tool-use +- arXiv categories: cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30690 diff --git a/papers/items/2026-2605-30711-sage-a-novelty-gate-for-efficient-memory-evolution-in-agentic-llms.md b/papers/items/2026-2605-30711-sage-a-novelty-gate-for-efficient-memory-evolution-in-agentic-llms.md new file mode 100644 index 0000000..41e7977 --- /dev/null +++ b/papers/items/2026-2605-30711-sage-a-novelty-gate-for-efficient-memory-evolution-in-agentic-llms.md @@ -0,0 +1,65 @@ +# Paper: SAGE: A Novelty Gate for Efficient Memory Evolution in Agentic LLMs + +--- +type: paper +title: "SAGE: A Novelty Gate for Efficient Memory Evolution in Agentic LLMs" +authors: Sijia Wang, Dhanajit Brahma, Ricardo Henao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30711 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.LG + - stat.ML +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.CL, cs.AI, cs.LG, stat.ML +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30711 diff --git a/papers/items/2026-2605-30858-forecastcompass-guiding-agentic-forecasting-with-adaptive-factor-memory.md b/papers/items/2026-2605-30858-forecastcompass-guiding-agentic-forecasting-with-adaptive-factor-memory.md new file mode 100644 index 0000000..6f16300 --- /dev/null +++ b/papers/items/2026-2605-30858-forecastcompass-guiding-agentic-forecasting-with-adaptive-factor-memory.md @@ -0,0 +1,63 @@ +# Paper: ForecastCompass: Guiding Agentic Forecasting with Adaptive Factor Memory + +--- +type: paper +title: "ForecastCompass: Guiding Agentic Forecasting with Adaptive Factor Memory" +authors: Yurui Chang, Yongkang Du, Yuanpu Cao, Jinghui Chen, Lu Lin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30858 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, rag, reasoning, tool-use +- arXiv categories: cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30858 diff --git a/papers/items/2026-2605-30883-trace-task-aware-adaptive-self-evolving-agentic-jailbreaking.md b/papers/items/2026-2605-30883-trace-task-aware-adaptive-self-evolving-agentic-jailbreaking.md new file mode 100644 index 0000000..278e265 --- /dev/null +++ b/papers/items/2026-2605-30883-trace-task-aware-adaptive-self-evolving-agentic-jailbreaking.md @@ -0,0 +1,65 @@ +# Paper: TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking + +--- +type: paper +title: "TRACE: Task-Aware Adaptive Self-Evolving Agentic Jailbreaking" +authors: Churui Zeng, Weiwei Qi, Kedong Xiu, Tianhang Zheng, Chaochao Lu, Liang He, Zhan Qin, Kui Ren +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30883 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.CR +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30883 diff --git a/papers/items/2026-2605-30907-bluefin-benchmarking-llm-agents-on-financial-spreadsheets.md b/papers/items/2026-2605-30907-bluefin-benchmarking-llm-agents-on-financial-spreadsheets.md new file mode 100644 index 0000000..08dc35f --- /dev/null +++ b/papers/items/2026-2605-30907-bluefin-benchmarking-llm-agents-on-financial-spreadsheets.md @@ -0,0 +1,62 @@ +# Paper: BlueFin: Benchmarking LLM Agents on Financial Spreadsheets + +--- +type: paper +title: "BlueFin: Benchmarking LLM Agents on Financial Spreadsheets" +authors: Srivatsa Kundurthy, Clara Na, Colton Moraine, Anoushka Mohta, Case Winter, George Fang, John Ling, Emma Strubell, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30907 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, rag +- arXiv categories: cs.SE, cs.AI, cs.CL, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30907 diff --git a/papers/items/2026-2605-30947-extending-ai-for-research-to-the-humanities-a-multi-agent-framework-for-evidence.md b/papers/items/2026-2605-30947-extending-ai-for-research-to-the-humanities-a-multi-agent-framework-for-evidence.md new file mode 100644 index 0000000..1290886 --- /dev/null +++ b/papers/items/2026-2605-30947-extending-ai-for-research-to-the-humanities-a-multi-agent-framework-for-evidence.md @@ -0,0 +1,62 @@ +# Paper: Extending AI for Research to the Humanities: A Multi-Agent Framework for Evidence-Grounded Scholarship + +--- +type: paper +title: "Extending AI for Research to the Humanities: A Multi-Agent Framework for Evidence-Grounded Scholarship" +authors: Yating Pan, Jiajun Zhang, Jun Wang, Qi Su +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.30947 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.30947 diff --git a/papers/items/2026-2605-31075-task-focused-memorization-for-multimodal-agents.md b/papers/items/2026-2605-31075-task-focused-memorization-for-multimodal-agents.md new file mode 100644 index 0000000..05d8647 --- /dev/null +++ b/papers/items/2026-2605-31075-task-focused-memorization-for-multimodal-agents.md @@ -0,0 +1,61 @@ +# Paper: Task-Focused Memorization for Multimodal Agents + +--- +type: paper +title: Task-Focused Memorization for Multimodal Agents +authors: Tao Zou, Yichen He, Tian Qiu, Yuan Lin, Hang Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.31075 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, embodied-agent, memory +- arXiv categories: cs.CV +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.31075 diff --git a/papers/items/2026-2605-31268-mellum2-technical-report.md b/papers/items/2026-2605-31268-mellum2-technical-report.md new file mode 100644 index 0000000..b344be8 --- /dev/null +++ b/papers/items/2026-2605-31268-mellum2-technical-report.md @@ -0,0 +1,62 @@ +# Paper: Mellum2 Technical Report + +--- +type: paper +title: Mellum2 Technical Report +authors: Marko Kojic, Ivan Bondyrev, Aral de Moor, Joseph Shtok, Petr Borovlev, Kseniia Lysaniuk, Madeeswaran Kannan, Ivan Dolgov, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.31268 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, agent-safety, coding-agent, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.31268 diff --git a/papers/items/2026-2605-31278-industrializing-prediction-powered-inference-the-glide-library-for-reliable-gena.md b/papers/items/2026-2605-31278-industrializing-prediction-powered-inference-the-glide-library-for-reliable-gena.md new file mode 100644 index 0000000..044722f --- /dev/null +++ b/papers/items/2026-2605-31278-industrializing-prediction-powered-inference-the-glide-library-for-reliable-gena.md @@ -0,0 +1,61 @@ +# Paper: Industrializing Prediction-Powered Inference: The GLIDE Library for Reliable GenAI and Agentic Systems Evaluation + +--- +type: paper +title: "Industrializing Prediction-Powered Inference: The GLIDE Library for Reliable GenAI and Agentic Systems Evaluation" +authors: Grégoire Martinon, Ibrahim Merad, Mohammed Raki +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.31278 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG + - stat.ME +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.AI, cs.LG, stat.ME +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.31278 diff --git a/papers/items/2026-2605-31308-tracegraph-shared-decision-landscapes-for-diagnosing-and-improving-agent-traject.md b/papers/items/2026-2605-31308-tracegraph-shared-decision-landscapes-for-diagnosing-and-improving-agent-traject.md new file mode 100644 index 0000000..d6049e1 --- /dev/null +++ b/papers/items/2026-2605-31308-tracegraph-shared-decision-landscapes-for-diagnosing-and-improving-agent-traject.md @@ -0,0 +1,62 @@ +# Paper: TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories + +--- +type: paper +title: "TraceGraph: Shared Decision Landscapes for Diagnosing and Improving Agent Trajectories" +authors: Junjie Nian, Kang Chen, Ge Zhang, Yixin Cao, Yugang Jiang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.31308 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - embodied-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, computer-use, embodied-agent, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.31308 diff --git a/papers/items/2026-2605-31377-dynatree-dynamic-agentic-retrieval-tree-for-time-sensitive-news-retrieval.md b/papers/items/2026-2605-31377-dynatree-dynamic-agentic-retrieval-tree-for-time-sensitive-news-retrieval.md new file mode 100644 index 0000000..d8ba424 --- /dev/null +++ b/papers/items/2026-2605-31377-dynatree-dynamic-agentic-retrieval-tree-for-time-sensitive-news-retrieval.md @@ -0,0 +1,63 @@ +# Paper: DynaTree: Dynamic Agentic Retrieval Tree for Time-Sensitive News Retrieval + +--- +type: paper +title: "DynaTree: Dynamic Agentic Retrieval Tree for Time-Sensitive News Retrieval" +authors: Siyuan Qi, Xinyuan Wang, Yingxuan Yang, Haochuan Guo, Jianghao Lin, Weiwen Liu, Yong Yu, Weinan Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2605.31377 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use +- arXiv categories: cs.IR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2605.31377 diff --git a/papers/items/2026-2606-00198-bagen-are-llm-agents-budget-aware.md b/papers/items/2026-2606-00198-bagen-are-llm-agents-budget-aware.md new file mode 100644 index 0000000..db31aa8 --- /dev/null +++ b/papers/items/2026-2606-00198-bagen-are-llm-agents-budget-aware.md @@ -0,0 +1,63 @@ +# Paper: BAGEN: Are LLM Agents Budget-Aware? + +--- +type: paper +title: "BAGEN: Are LLM Agents Budget-Aware?" +authors: Yuxiang Lin, Zihan Wang, Mengyang Liu, Yuxuan Shan, Longju Bai, Junyao Zhang, Xing Jin, Boshan Chen, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00198 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, tool-use +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00198 diff --git a/papers/items/2026-2606-00341-rogue-misaligned-agent-behavior-arising-from-ordinary-computer-use.md b/papers/items/2026-2606-00341-rogue-misaligned-agent-behavior-arising-from-ordinary-computer-use.md new file mode 100644 index 0000000..ce9d712 --- /dev/null +++ b/papers/items/2026-2606-00341-rogue-misaligned-agent-behavior-arising-from-ordinary-computer-use.md @@ -0,0 +1,63 @@ +# Paper: ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use + +--- +type: paper +title: "ROGUE: Misaligned Agent Behavior Arising from Ordinary Computer Use" +authors: Jeremy Tien, Abishek Anand, Yu-Rou Tuan, Yuchen Shen, J. Zico Kolter, Aran Nayebi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00341 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, tool-use, workflow-agent +- arXiv categories: cs.LG, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00341 diff --git a/papers/items/2026-2606-00610-memgraphrag-memory-based-multi-agent-system-for-graph-retrieval-augmented-genera.md b/papers/items/2026-2606-00610-memgraphrag-memory-based-multi-agent-system-for-graph-retrieval-augmented-genera.md new file mode 100644 index 0000000..4d3552d --- /dev/null +++ b/papers/items/2026-2606-00610-memgraphrag-memory-based-multi-agent-system-for-graph-retrieval-augmented-genera.md @@ -0,0 +1,65 @@ +# Paper: MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation + +--- +type: paper +title: "MemGraphRAG: Memory-based Multi-Agent System for Graph Retrieval-Augmented Generation" +authors: Chuanjie Wu, Zhishang Xiang, Yunbo Tang, Zerui Chen, Qinggang Zhang, Jinsong Su +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00610 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-30 +updated_at: 2026-05-30 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, memory, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.IR, cs.AI, cs.MA +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00610 diff --git a/papers/items/2026-2606-00611-trace-trajectory-risk-aware-compression-for-long-horizon-agent-safety.md b/papers/items/2026-2606-00611-trace-trajectory-risk-aware-compression-for-long-horizon-agent-safety.md new file mode 100644 index 0000000..980fd6a --- /dev/null +++ b/papers/items/2026-2606-00611-trace-trajectory-risk-aware-compression-for-long-horizon-agent-safety.md @@ -0,0 +1,60 @@ +# Paper: TRACE: Trajectory Risk-Aware Compression for Long-Horizon Agent Safety + +--- +type: paper +title: "TRACE: Trajectory Risk-Aware Compression for Long-Horizon Agent Safety" +authors: Zhepei Hong, Lin Wang, Liting Li, Haokai Ma, Junfeng Fang, Fei Shen, Dan Zhang, Xiang Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00611 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-30 +updated_at: 2026-05-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, planning +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00611 diff --git a/papers/items/2026-2606-00619-mempro-agentic-memory-systems-as-evolvable-programs.md b/papers/items/2026-2606-00619-mempro-agentic-memory-systems-as-evolvable-programs.md new file mode 100644 index 0000000..da627d2 --- /dev/null +++ b/papers/items/2026-2606-00619-mempro-agentic-memory-systems-as-evolvable-programs.md @@ -0,0 +1,63 @@ +# Paper: MemPro: Agentic Memory Systems as Evolvable Programs + +--- +type: paper +title: "MemPro: Agentic Memory Systems as Evolvable Programs" +authors: Qingshan Liu, Guoqing Wang, Wen Wu, Jingqi Huang, Xinqi Tao, Dejia Song, Jie Zhou, Liang He +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00619 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-30 +updated_at: 2026-05-30 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, planning, rag +- arXiv categories: cs.CL, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00619 diff --git a/papers/items/2026-2606-00644-foresci-evaluating-llm-agents-for-forward-looking-ai-research-judgment.md b/papers/items/2026-2606-00644-foresci-evaluating-llm-agents-for-forward-looking-ai-research-judgment.md new file mode 100644 index 0000000..e9338b6 --- /dev/null +++ b/papers/items/2026-2606-00644-foresci-evaluating-llm-agents-for-forward-looking-ai-research-judgment.md @@ -0,0 +1,59 @@ +# Paper: ForeSci: Evaluating LLM Agents for Forward-Looking AI Research Judgment + +--- +type: paper +title: "ForeSci: Evaluating LLM Agents for Forward-Looking AI Research Judgment" +authors: Qiuyu Tian, Haojie Yin, Yingce Xia, Youyong Kong, Zequn Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00644 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-30 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, rag +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00644 diff --git a/papers/items/2026-2606-00756-comic-collaborative-memory-and-insights-circulation-for-long-horizon-llm-agents-.md b/papers/items/2026-2606-00756-comic-collaborative-memory-and-insights-circulation-for-long-horizon-llm-agents-.md new file mode 100644 index 0000000..3fb5028 --- /dev/null +++ b/papers/items/2026-2606-00756-comic-collaborative-memory-and-insights-circulation-for-long-horizon-llm-agents-.md @@ -0,0 +1,64 @@ +# Paper: CoMIC: Collaborative Memory and Insights Circulation for Long-Horizon LLM Agents in Cloud-Edge Systems + +--- +type: paper +title: "CoMIC: Collaborative Memory and Insights Circulation for Long-Horizon LLM Agents in Cloud-Edge Systems" +authors: Yannan Wang, Longli Yang, Zhen Liu, Abhishek Kumar, Carsten Maple +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00756 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-30 +updated_at: 2026-05-30 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00756 diff --git a/papers/items/2026-2606-00914-adversarial-feeds-steer-llm-agent-decisions-against-their-defaults.md b/papers/items/2026-2606-00914-adversarial-feeds-steer-llm-agent-decisions-against-their-defaults.md new file mode 100644 index 0000000..594dd89 --- /dev/null +++ b/papers/items/2026-2606-00914-adversarial-feeds-steer-llm-agent-decisions-against-their-defaults.md @@ -0,0 +1,63 @@ +# Paper: Adversarial Feeds Steer LLM Agent Decisions Against Their Defaults + +--- +type: paper +title: Adversarial Feeds Steer LLM Agent Decisions Against Their Defaults +authors: Rana Muhammad Usman +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00914 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-30 +updated_at: 2026-05-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.AI, cs.CL, cs.CR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00914 diff --git a/papers/items/2026-2606-00915-autonomous-agentic-design-for-photonics.md b/papers/items/2026-2606-00915-autonomous-agentic-design-for-photonics.md new file mode 100644 index 0000000..c18c6f1 --- /dev/null +++ b/papers/items/2026-2606-00915-autonomous-agentic-design-for-photonics.md @@ -0,0 +1,61 @@ +# Paper: Autonomous agentic design for photonics + +--- +type: paper +title: Autonomous agentic design for photonics +authors: Prashanta Kharel, Amin Khavasi, Xinzhong Chen, Tyler W. Hughes +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00915 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-30 +updated_at: 2026-05-30 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - physics.optics +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, computer-use, tool-use, world-model +- arXiv categories: physics.optics +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00915 diff --git a/papers/items/2026-2606-00922-a-machine-to-machine-knowledge-guided-llm-agent-for-generalizable-radiotherapy-t.md b/papers/items/2026-2606-00922-a-machine-to-machine-knowledge-guided-llm-agent-for-generalizable-radiotherapy-t.md new file mode 100644 index 0000000..229ec03 --- /dev/null +++ b/papers/items/2026-2606-00922-a-machine-to-machine-knowledge-guided-llm-agent-for-generalizable-radiotherapy-t.md @@ -0,0 +1,63 @@ +# Paper: A Machine-to-Machine Knowledge-Guided LLM Agent for Generalizable Radiotherapy Treatment Planning + +--- +type: paper +title: A Machine-to-Machine Knowledge-Guided LLM Agent for Generalizable Radiotherapy Treatment Planning +authors: Md Mainul Abrar, Xun Jia, Yujie Chi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00922 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-30 +updated_at: 2026-05-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - planning + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - physics.med-ph + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, planning, reasoning +- arXiv categories: physics.med-ph, cs.RO +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00922 diff --git a/papers/items/2026-2606-00939-fincom-a-financial-multi-agent-demo-with-disagree-or-commit-deliberation.md b/papers/items/2026-2606-00939-fincom-a-financial-multi-agent-demo-with-disagree-or-commit-deliberation.md new file mode 100644 index 0000000..c73cd08 --- /dev/null +++ b/papers/items/2026-2606-00939-fincom-a-financial-multi-agent-demo-with-disagree-or-commit-deliberation.md @@ -0,0 +1,63 @@ +# Paper: FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation + +--- +type: paper +title: "FinCom: A Financial Multi-Agent Demo with Disagree-or-Commit Deliberation" +authors: Chao Peter Yang, Zixiao Tan, Kaisen Yao, Ziyu Zhou, Eleanor Jiang, Michael Wu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.00939 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-31 +updated_at: 2026-05-31 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.MA +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.00939 diff --git a/papers/items/2026-2606-01041-expweaver-llm-agents-learn-from-experience-via-latent-rag.md b/papers/items/2026-2606-01041-expweaver-llm-agents-learn-from-experience-via-latent-rag.md new file mode 100644 index 0000000..fa572dc --- /dev/null +++ b/papers/items/2026-2606-01041-expweaver-llm-agents-learn-from-experience-via-latent-rag.md @@ -0,0 +1,63 @@ +# Paper: ExpWeaver: LLM Agents Learn from Experience via Latent RAG + +--- +type: paper +title: "ExpWeaver: LLM Agents Learn from Experience via Latent RAG" +authors: Tao Feng, Tianyang Luo, Jingjun Xu, Zhigang Hua, Yan Xie, Shuang Yang, Ge Liu, Jiaxuan You +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01041 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-31 +updated_at: 2026-05-31 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent, rag-agent +- inferred topics: agent-evaluation, coding-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01041 diff --git a/papers/items/2026-2606-01138-memorywire-a-vendor-neutral-wire-format-for-agent-memory-operations.md b/papers/items/2026-2606-01138-memorywire-a-vendor-neutral-wire-format-for-agent-memory-operations.md new file mode 100644 index 0000000..5f5343a --- /dev/null +++ b/papers/items/2026-2606-01138-memorywire-a-vendor-neutral-wire-format-for-agent-memory-operations.md @@ -0,0 +1,63 @@ +# Paper: memorywire: A Vendor-Neutral Wire Format for Agent Memory Operations + +--- +type: paper +title: "memorywire: A Vendor-Neutral Wire Format for Agent Memory Operations" +authors: Thamilvendhan Munirathinam +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01138 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-31 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.DC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, workflow-agent +- arXiv categories: cs.CR, cs.AI, cs.DC +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01138 diff --git a/papers/items/2026-2606-01166-braveguard-from-open-world-threats-to-safer-computer-use-agents.md b/papers/items/2026-2606-01166-braveguard-from-open-world-threats-to-safer-computer-use-agents.md new file mode 100644 index 0000000..0dc6d21 --- /dev/null +++ b/papers/items/2026-2606-01166-braveguard-from-open-world-threats-to-safer-computer-use-agents.md @@ -0,0 +1,63 @@ +# Paper: BraveGuard: From Open-World Threats to Safer Computer-Use Agents + +--- +type: paper +title: "BraveGuard: From Open-World Threats to Safer Computer-Use Agents" +authors: Yunhao Feng, Xiaohu Du, Xinhao Deng, Yifan Ding, Ming Wen, Yixu Wang, Yuxiang Xie, Baihui Zheng, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01166 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-31 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, rag, tool-use +- arXiv categories: cs.CR, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01166 diff --git a/papers/items/2026-2606-01185-skill-issues-data-centric-optimization-of-lakehouse-agents.md b/papers/items/2026-2606-01185-skill-issues-data-centric-optimization-of-lakehouse-agents.md new file mode 100644 index 0000000..615662b --- /dev/null +++ b/papers/items/2026-2606-01185-skill-issues-data-centric-optimization-of-lakehouse-agents.md @@ -0,0 +1,63 @@ +# Paper: "Skill issues'': data-centric optimization of lakehouse agents + +--- +type: paper +title: "\"Skill issues'': data-centric optimization of lakehouse agents" +authors: Nicole Rose Schneider, Davide Ghilardi, Giacomo Piccinini, Jacopo Tagliabue +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01185 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-31 +updated_at: 2026-05-31 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01185 diff --git a/papers/items/2026-2606-01199-can-llm-agents-sustain-long-horizon-organizational-dynamics.md b/papers/items/2026-2606-01199-can-llm-agents-sustain-long-horizon-organizational-dynamics.md new file mode 100644 index 0000000..2bec0ce --- /dev/null +++ b/papers/items/2026-2606-01199-can-llm-agents-sustain-long-horizon-organizational-dynamics.md @@ -0,0 +1,64 @@ +# Paper: Can LLM Agents Sustain Long-Horizon Organizational Dynamics? + +--- +type: paper +title: Can LLM Agents Sustain Long-Horizon Organizational Dynamics? +authors: Xuancheng Zhu, Yang Yue, Shuaibing Wan, Zihan Dou, Xiaohan Zhang, Yongrui Liu, Guoshun Nan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01199 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-31 +updated_at: 2026-05-31 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - planning + - rag + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: language-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent, planning-agent +- inferred topics: agent-evaluation, memory, multi-agent, planning, rag, workflow-agent, world-model +- arXiv categories: cs.AI +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01199 diff --git a/papers/items/2026-2606-01222-rag-driven-multi-agent-llm-framework-with-task-decomposition-for-beyond-5g-auto-.md b/papers/items/2026-2606-01222-rag-driven-multi-agent-llm-framework-with-task-decomposition-for-beyond-5g-auto-.md new file mode 100644 index 0000000..b3d39e3 --- /dev/null +++ b/papers/items/2026-2606-01222-rag-driven-multi-agent-llm-framework-with-task-decomposition-for-beyond-5g-auto-.md @@ -0,0 +1,62 @@ +# Paper: RAG-driven Multi-Agent LLM Framework with Task Decomposition for Beyond 5G Auto-Configuration + +--- +type: paper +title: RAG-driven Multi-Agent LLM Framework with Task Decomposition for Beyond 5G Auto-Configuration +authors: İrşat Emin Sarıdaş, Onur Salan, Ali Görçin, Ibrahim Hokelek, Hakan Ali Çırpan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01222 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-31 +updated_at: 2026-05-31 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SP +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, multi-agent, planning, rag, reasoning +- arXiv categories: eess.SP +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01222 diff --git a/papers/items/2026-2606-01385-bridging-requirements-and-architecture-multi-agent-orchestration-with-external-k.md b/papers/items/2026-2606-01385-bridging-requirements-and-architecture-multi-agent-orchestration-with-external-k.md new file mode 100644 index 0000000..6864f29 --- /dev/null +++ b/papers/items/2026-2606-01385-bridging-requirements-and-architecture-multi-agent-orchestration-with-external-k.md @@ -0,0 +1,65 @@ +# Paper: Bridging Requirements and Architecture: Multi-Agent Orchestration with External Knowledge and Hierarchical Memory + +--- +type: paper +title: "Bridging Requirements and Architecture: Multi-Agent Orchestration with External Knowledge and Hierarchical Memory" +authors: Ruiyin Li, Yiran Zhang, Xiyu Zhou, Yangxiao Cai, Peng Liang, Weisong Sun, Jifeng Xuan, Zhi Jin, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01385 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-31 +updated_at: 2026-05-31 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - multi-agent + - rag + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, coding-agent, memory, multi-agent, rag, reasoning, workflow-agent +- arXiv categories: cs.SE, cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01385 diff --git a/papers/items/2026-2606-01416-self-healing-agentic-orchestrators-for-reliable-tool-augmented-large-language-mo.md b/papers/items/2026-2606-01416-self-healing-agentic-orchestrators-for-reliable-tool-augmented-large-language-mo.md new file mode 100644 index 0000000..0898ee6 --- /dev/null +++ b/papers/items/2026-2606-01416-self-healing-agentic-orchestrators-for-reliable-tool-augmented-large-language-mo.md @@ -0,0 +1,65 @@ +# Paper: Self-Healing Agentic Orchestrators for Reliable Tool-Augmented Large Language Model Systems + +--- +type: paper +title: Self-Healing Agentic Orchestrators for Reliable Tool-Augmented Large Language Model Systems +authors: Rahul Suresh Babu, Adarsh Agrawal +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01416 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-31 +updated_at: 2026-05-31 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01416 diff --git a/papers/items/2026-2606-01528-joint-agent-memory-and-exploration-learning-via-novelty-signals.md b/papers/items/2026-2606-01528-joint-agent-memory-and-exploration-learning-via-novelty-signals.md new file mode 100644 index 0000000..17d40f9 --- /dev/null +++ b/papers/items/2026-2606-01528-joint-agent-memory-and-exploration-learning-via-novelty-signals.md @@ -0,0 +1,62 @@ +# Paper: Joint Agent Memory and Exploration Learning via Novelty Signals + +--- +type: paper +title: Joint Agent Memory and Exploration Learning via Novelty Signals +authors: Shizuo Tian, Xiaohong Weng, Rui Kong, Yuxuan Chen, Guohong Liu, Yuebing Song, Jiacheng Liu, Yuchen Li, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01528 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, rag, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01528 diff --git a/papers/items/2026-2606-01613-techrag-evidence-gated-multimodal-agentic-rag-for-technical-literature-reasoning.md b/papers/items/2026-2606-01613-techrag-evidence-gated-multimodal-agentic-rag-for-technical-literature-reasoning.md new file mode 100644 index 0000000..ee26931 --- /dev/null +++ b/papers/items/2026-2606-01613-techrag-evidence-gated-multimodal-agentic-rag-for-technical-literature-reasoning.md @@ -0,0 +1,67 @@ +# Paper: TechRAG: Evidence-Gated Multimodal Agentic RAG for Technical Literature Reasoning + +--- +type: paper +title: "TechRAG: Evidence-Gated Multimodal Agentic RAG for Technical Literature Reasoning" +authors: Kanwar Bharat Singh +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01613 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-13 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - multi-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, multi-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.IR, cs.AI, cs.MA +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01613 diff --git a/papers/items/2026-2606-01815-crab-bench-evaluating-llm-agents-under-complex-task-dependencies-and-human-align.md b/papers/items/2026-2606-01815-crab-bench-evaluating-llm-agents-under-complex-task-dependencies-and-human-align.md new file mode 100644 index 0000000..77a63b5 --- /dev/null +++ b/papers/items/2026-2606-01815-crab-bench-evaluating-llm-agents-under-complex-task-dependencies-and-human-align.md @@ -0,0 +1,60 @@ +# Paper: CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation + +--- +type: paper +title: "CRAB-Bench: Evaluating LLM Agents under Complex Task Dependencies and Human-aligned User Simulation" +authors: Danqing Wang, Akshay Sivaraman, Lei Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01815 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, tool-use, world-model +- arXiv categories: cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01815 diff --git a/papers/items/2026-2606-01961-automedbench-towards-medical-autoresearch-with-agentic-ai-models.md b/papers/items/2026-2606-01961-automedbench-towards-medical-autoresearch-with-agentic-ai-models.md new file mode 100644 index 0000000..42b53c3 --- /dev/null +++ b/papers/items/2026-2606-01961-automedbench-towards-medical-autoresearch-with-agentic-ai-models.md @@ -0,0 +1,61 @@ +# Paper: AutoMedBench: Towards Medical AutoResearch with Agentic AI Models + +--- +type: paper +title: "AutoMedBench: Towards Medical AutoResearch with Agentic AI Models" +authors: Junqi Liu, Selena Song, Yuhan Wang, Jiawei Mao, Hardy Chen, Xiaoke Huang, Tianhao Qi, Pengfei Guo, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.01961 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, planning, rag, workflow-agent +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.01961 diff --git a/papers/items/2026-2606-02109-badger-bridging-agentic-and-deterministic-evaluation-for-generative-enterprise-r.md b/papers/items/2026-2606-02109-badger-bridging-agentic-and-deterministic-evaluation-for-generative-enterprise-r.md new file mode 100644 index 0000000..888206d --- /dev/null +++ b/papers/items/2026-2606-02109-badger-bridging-agentic-and-deterministic-evaluation-for-generative-enterprise-r.md @@ -0,0 +1,63 @@ +# Paper: BADGER: Bridging Agentic and Deterministic Evaluation for Generative Enterprise Reasoning + +--- +type: paper +title: "BADGER: Bridging Agentic and Deterministic Evaluation for Generative Enterprise Reasoning" +authors: Shannon Serrao, Soumitra Chatterjee, Dorina Strori, Abhishek Sharma, Nathan Miller +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02109 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02109 diff --git a/papers/items/2026-2606-02302-seclaw-spec-driven-security-task-synthesis-for-evaluating-autonomous-agents.md b/papers/items/2026-2606-02302-seclaw-spec-driven-security-task-synthesis-for-evaluating-autonomous-agents.md new file mode 100644 index 0000000..0a50009 --- /dev/null +++ b/papers/items/2026-2606-02302-seclaw-spec-driven-security-task-synthesis-for-evaluating-autonomous-agents.md @@ -0,0 +1,64 @@ +# Paper: SeClaw: Spec-Driven Security Task Synthesis for Evaluating Autonomous Agents + +--- +type: paper +title: "SeClaw: Spec-Driven Security Task Synthesis for Evaluating Autonomous Agents" +authors: Hao Cheng, Changtao Miao, Tianle Song, Yin Wu, He Liu, Erjia Xiao, Junchi Chen, Xiaoyu Shi, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02302 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-safety, autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02302 diff --git a/papers/items/2026-2606-02372-comap-co-evolving-world-models-and-agent-policies-for-llm-agents.md b/papers/items/2026-2606-02372-comap-co-evolving-world-models-and-agent-policies-for-llm-agents.md new file mode 100644 index 0000000..b4414f5 --- /dev/null +++ b/papers/items/2026-2606-02372-comap-co-evolving-world-models-and-agent-policies-for-llm-agents.md @@ -0,0 +1,64 @@ +# Paper: COMAP: Co-Evolving World Models and Agent Policies for LLM Agents + +--- +type: paper +title: "COMAP: Co-Evolving World Models and Agent Policies for LLM Agents" +authors: Youwei Liu, Jian Wang, Hanlin Wang, Wenjie Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02372 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - planning + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent, planning-agent +- inferred topics: agent-evaluation, embodied-agent, planning, reasoning, tool-use, world-model +- arXiv categories: cs.AI, cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02372 diff --git a/papers/items/2026-2606-02380-spade-bench-evaluating-spontaneous-strategic-deception-in-agents-via-plan-action.md b/papers/items/2026-2606-02380-spade-bench-evaluating-spontaneous-strategic-deception-in-agents-via-plan-action.md new file mode 100644 index 0000000..f19eafe --- /dev/null +++ b/papers/items/2026-2606-02380-spade-bench-evaluating-spontaneous-strategic-deception-in-agents-via-plan-action.md @@ -0,0 +1,63 @@ +# Paper: SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence + +--- +type: paper +title: "SPADE-Bench: Evaluating Spontaneous Strategic Deception in Agents via Plan-Action Divergence" +authors: Yuyan Bu, Haowei Li, Qirui Zheng, Bowen Dong, Kaiyue Yang, Jiaming Ji, Yingshui Tan, Wenxin Li, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02380 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use, planning, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02380 diff --git a/papers/items/2026-2606-02388-policy-and-world-modeling-co-training-for-language-agents.md b/papers/items/2026-2606-02388-policy-and-world-modeling-co-training-for-language-agents.md new file mode 100644 index 0000000..26e8e90 --- /dev/null +++ b/papers/items/2026-2606-02388-policy-and-world-modeling-co-training-for-language-agents.md @@ -0,0 +1,61 @@ +# Paper: Policy and World Modeling Co-Training for Language Agents + +--- +type: paper +title: Policy and World Modeling Co-Training for Language Agents +authors: Ning Lu, Baijiong Lin, Shengcai Liu, Jiahao Wu, Haoze Lv, Yanbin Wei, Lingting Zhu, Shengju Qian, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02388 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, tool-use, world-model +- arXiv categories: cs.LG, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02388 diff --git a/papers/items/2026-2606-02404-k-browsecomp-a-web-browsing-agent-benchmark-grounded-in-korean-contexts.md b/papers/items/2026-2606-02404-k-browsecomp-a-web-browsing-agent-benchmark-grounded-in-korean-contexts.md new file mode 100644 index 0000000..3d4e911 --- /dev/null +++ b/papers/items/2026-2606-02404-k-browsecomp-a-web-browsing-agent-benchmark-grounded-in-korean-contexts.md @@ -0,0 +1,59 @@ +# Paper: K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts + +--- +type: paper +title: "K-BrowseComp: A Web Browsing Agent Benchmark Grounded in Korean Contexts" +authors: Nahyun Lee, Dongkeun Yoon, Guijin Son, Geewook Kim, Dayoon Ko, Jeonghun Park, Haneul Yoo, Jaewon Cho, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02404 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, reasoning +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02404 diff --git a/papers/items/2026-2606-02461-agentcl-toward-rigorous-evaluation-of-continual-learning-in-language-agents.md b/papers/items/2026-2606-02461-agentcl-toward-rigorous-evaluation-of-continual-learning-in-language-agents.md new file mode 100644 index 0000000..df889da --- /dev/null +++ b/papers/items/2026-2606-02461-agentcl-toward-rigorous-evaluation-of-continual-learning-in-language-agents.md @@ -0,0 +1,66 @@ +# Paper: AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents + +--- +type: paper +title: "AgentCL: Toward Rigorous Evaluation of Continual Learning in Language Agents" +authors: Yiheng Shu, Bernal Jiménez Gutiérrez, Saisri Padmaja Jonnalagedda, Yuguang Yao, Huan Sun, Yu Su +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02461 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - memory + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, memory, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CL +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02461 diff --git a/papers/items/2026-2606-02497-bridging-the-last-mile-of-time-series-forecasting-with-llm-agents.md b/papers/items/2026-2606-02497-bridging-the-last-mile-of-time-series-forecasting-with-llm-agents.md new file mode 100644 index 0000000..58fe290 --- /dev/null +++ b/papers/items/2026-2606-02497-bridging-the-last-mile-of-time-series-forecasting-with-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: Bridging the Last Mile of Time Series Forecasting with LLM Agents + +--- +type: paper +title: Bridging the Last Mile of Time Series Forecasting with LLM Agents +authors: Yuhua Liao, Zetian Wang, Qiangqiang Nie, Zhenhua Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02497 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-safety + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-safety, memory, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02497 diff --git a/papers/items/2026-2606-02812-traj-evolve-a-self-evolving-multi-agent-system-for-patient-trajectory-modeling-i.md b/papers/items/2026-2606-02812-traj-evolve-a-self-evolving-multi-agent-system-for-patient-trajectory-modeling-i.md new file mode 100644 index 0000000..7bca052 --- /dev/null +++ b/papers/items/2026-2606-02812-traj-evolve-a-self-evolving-multi-agent-system-for-patient-trajectory-modeling-i.md @@ -0,0 +1,64 @@ +# Paper: Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection + +--- +type: paper +title: "Traj-Evolve: A Self-Evolving Multi-Agent System for Patient Trajectory Modeling in Lung Cancer Early Detection" +authors: Sihang Zeng, Matthew Thompson, Ruth Etzioni, Meliha Yetisgen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02812 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - multi-agent + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, memory, multi-agent, rag, reasoning +- arXiv categories: cs.AI, cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02812 diff --git a/papers/items/2026-2606-02965-what-benchmarks-don-t-measure-the-case-for-evaluating-abstention-competence-in-a.md b/papers/items/2026-2606-02965-what-benchmarks-don-t-measure-the-case-for-evaluating-abstention-competence-in-a.md new file mode 100644 index 0000000..83be11b --- /dev/null +++ b/papers/items/2026-2606-02965-what-benchmarks-don-t-measure-the-case-for-evaluating-abstention-competence-in-a.md @@ -0,0 +1,62 @@ +# Paper: What Benchmarks Don't Measure: The Case for Evaluating Abstention Competence in Autonomous Agents + +--- +type: paper +title: "What Benchmarks Don't Measure: The Case for Evaluating Abstention Competence in Autonomous Agents" +authors: Victor Ojewale, Suresh Venkatasubramanian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.02965 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, computer-use, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.02965 diff --git a/papers/items/2026-2606-03108-evotrainer-co-evolving-llm-policies-and-training-harnesses-for-autonomous-agenti.md b/papers/items/2026-2606-03108-evotrainer-co-evolving-llm-policies-and-training-harnesses-for-autonomous-agenti.md new file mode 100644 index 0000000..b7fccf1 --- /dev/null +++ b/papers/items/2026-2606-03108-evotrainer-co-evolving-llm-policies-and-training-harnesses-for-autonomous-agenti.md @@ -0,0 +1,61 @@ +# Paper: EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning + +--- +type: paper +title: "EvoTrainer: Co-Evolving LLM Policies and Training Harnesses for Autonomous Agentic Reinforcement Learning" +authors: Guhong Chen, Yingcheng Shi, Yongbin Li, Binhua Li, Xander Xu, Hu Wei, Shiwen Ni, Min Yang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.03108 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, coding-agent, planning, reasoning +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.03108 diff --git a/papers/items/2026-2606-03135-uncertainty-aware-clarification-in-llm-agents-with-information-gain.md b/papers/items/2026-2606-03135-uncertainty-aware-clarification-in-llm-agents-with-information-gain.md new file mode 100644 index 0000000..30e1f7d --- /dev/null +++ b/papers/items/2026-2606-03135-uncertainty-aware-clarification-in-llm-agents-with-information-gain.md @@ -0,0 +1,61 @@ +# Paper: Uncertainty-Aware Clarification in LLM Agents with Information Gain + +--- +type: paper +title: Uncertainty-Aware Clarification in LLM Agents with Information Gain +authors: Mengyi Deng, Zhiwei Li, Xin Li, Tingyu Zhu, Ying Zhao, Zhijiang Guo, Wei Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.03135 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, computer-use, rag, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.03135 diff --git a/papers/items/2026-2606-03157-clinicalmc-a-benchmark-for-multi-course-clinical-decision-making-with-large-lang.md b/papers/items/2026-2606-03157-clinicalmc-a-benchmark-for-multi-course-clinical-decision-making-with-large-lang.md new file mode 100644 index 0000000..d0025b2 --- /dev/null +++ b/papers/items/2026-2606-03157-clinicalmc-a-benchmark-for-multi-course-clinical-decision-making-with-large-lang.md @@ -0,0 +1,60 @@ +# Paper: ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models + +--- +type: paper +title: "ClinicalMC: A Benchmark for Multi-Course Clinical Decision-Making with Large Language Models" +authors: Ruihui Hou, Siyi Zhu, Ziyue Huai, Guangya Yu, Yongqi Fan, Chunming Wang, Tong Ruan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.03157 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, multi-agent, rag +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.03157 diff --git a/papers/items/2026-2606-03197-memtrain-self-supervised-context-memory-training.md b/papers/items/2026-2606-03197-memtrain-self-supervised-context-memory-training.md new file mode 100644 index 0000000..6184812 --- /dev/null +++ b/papers/items/2026-2606-03197-memtrain-self-supervised-context-memory-training.md @@ -0,0 +1,63 @@ +# Paper: MemTrain: Self-Supervised Context Memory Training + +--- +type: paper +title: "MemTrain: Self-Supervised Context Memory Training" +authors: Ziheng Li, Xingrun Xing, Haoqing Wang, Zhi-Hong Deng, Yehui Tang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.03197 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.03197 diff --git a/papers/items/2026-2606-03329-infomem-training-long-context-memory-agents-with-answer-conditioned-information-.md b/papers/items/2026-2606-03329-infomem-training-long-context-memory-agents-with-answer-conditioned-information-.md new file mode 100644 index 0000000..1be7e73 --- /dev/null +++ b/papers/items/2026-2606-03329-infomem-training-long-context-memory-agents-with-answer-conditioned-information-.md @@ -0,0 +1,61 @@ +# Paper: InfoMem: Training Long-Context Memory Agents with Answer-Conditioned Information Gain + +--- +type: paper +title: "InfoMem: Training Long-Context Memory Agents with Answer-Conditioned Information Gain" +authors: Tiancheng Han, Yong Li, Wuzhou Yu, Qiaosheng Zhang, Wenqi Shao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.03329 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.03329 diff --git a/papers/items/2026-2606-03374-emem-a-hybrid-spatio-temporal-memory-system-for-embodied-agents.md b/papers/items/2026-2606-03374-emem-a-hybrid-spatio-temporal-memory-system-for-embodied-agents.md new file mode 100644 index 0000000..0b422d9 --- /dev/null +++ b/papers/items/2026-2606-03374-emem-a-hybrid-spatio-temporal-memory-system-for-embodied-agents.md @@ -0,0 +1,63 @@ +# Paper: eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents + +--- +type: paper +title: "eMEM: A Hybrid Spatio-Temporal Memory System For Embodied Agents" +authors: A. Haroon Rasheed, Maria Kabtoul +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.03374 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, rag-agent +- inferred topics: agent-evaluation, embodied-agent, memory, planning, rag, tool-use +- arXiv categories: cs.RO +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.03374 diff --git a/papers/items/2026-2606-03544-sage-a-quantitative-evaluation-of-socialized-evolution-in-agent-ecosystems.md b/papers/items/2026-2606-03544-sage-a-quantitative-evaluation-of-socialized-evolution-in-agent-ecosystems.md new file mode 100644 index 0000000..9e8f29c --- /dev/null +++ b/papers/items/2026-2606-03544-sage-a-quantitative-evaluation-of-socialized-evolution-in-agent-ecosystems.md @@ -0,0 +1,62 @@ +# Paper: SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems + +--- +type: paper +title: "SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems" +authors: Linyue Pan, Yaoming Zhu, Lin Qiu, Xuezhi Cao, Xunliang Cai +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.03544 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.03544 diff --git a/papers/items/2026-2606-03657-diagnosing-knowledge-gaps-in-llm-tool-use-an-agentic-benchmark-for-novel-api-acq.md b/papers/items/2026-2606-03657-diagnosing-knowledge-gaps-in-llm-tool-use-an-agentic-benchmark-for-novel-api-acq.md new file mode 100644 index 0000000..327b146 --- /dev/null +++ b/papers/items/2026-2606-03657-diagnosing-knowledge-gaps-in-llm-tool-use-an-agentic-benchmark-for-novel-api-acq.md @@ -0,0 +1,61 @@ +# Paper: Diagnosing Knowledge Gaps in LLM Tool Use: An Agentic Benchmark for Novel API Acquisition + +--- +type: paper +title: "Diagnosing Knowledge Gaps in LLM Tool Use: An Agentic Benchmark for Novel API Acquisition" +authors: Jinnuo Liu, Yue Peng, Jinhan Niu, Hongyi Wen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.03657 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, rag, tool-use +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.03657 diff --git a/papers/items/2026-2606-03895-agent-libos-a-runtime-substrate-for-capability-controlled-self-evolving-llm-agen.md b/papers/items/2026-2606-03895-agent-libos-a-runtime-substrate-for-capability-controlled-self-evolving-llm-agen.md new file mode 100644 index 0000000..1edc35d --- /dev/null +++ b/papers/items/2026-2606-03895-agent-libos-a-runtime-substrate-for-capability-controlled-self-evolving-llm-agen.md @@ -0,0 +1,64 @@ +# Paper: Agent libOS: A Runtime Substrate for Capability-Controlled Self-Evolving LLM Agents + +--- +type: paper +title: "Agent libOS: A Runtime Substrate for Capability-Controlled Self-Evolving LLM Agents" +authors: Yingqi Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.03895 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.OS + - cs.AI + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, memory, planning, tool-use +- arXiv categories: cs.OS, cs.AI, cs.CR +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.03895 diff --git a/papers/items/2026-2606-04051-rubas-rubric-based-reinforcement-learning-for-agent-safety.md b/papers/items/2026-2606-04051-rubas-rubric-based-reinforcement-learning-for-agent-safety.md new file mode 100644 index 0000000..88bb4d8 --- /dev/null +++ b/papers/items/2026-2606-04051-rubas-rubric-based-reinforcement-learning-for-agent-safety.md @@ -0,0 +1,62 @@ +# Paper: RUBAS: Rubric-Based Reinforcement Learning for Agent Safety + +--- +type: paper +title: "RUBAS: Rubric-Based Reinforcement Learning for Agent Safety" +authors: Xian Qi Loye, Qinglin Su, Zhexin Zhang, Shiyao Cui, Qi Zhu, Fei Mi, Hongning Wang, Minlie Huang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04051 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.LG, cs.AI, cs.CR +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04051 diff --git a/papers/items/2026-2606-04120-salimory-orchestrating-cognitive-memory-for-conversational-agents.md b/papers/items/2026-2606-04120-salimory-orchestrating-cognitive-memory-for-conversational-agents.md new file mode 100644 index 0000000..6a3947d --- /dev/null +++ b/papers/items/2026-2606-04120-salimory-orchestrating-cognitive-memory-for-conversational-agents.md @@ -0,0 +1,63 @@ +# Paper: SaliMory: Orchestrating Cognitive Memory for Conversational Agents + +--- +type: paper +title: "SaliMory: Orchestrating Cognitive Memory for Conversational Agents" +authors: Kai Zhang, Xinyuan Zhang, Hongda Jiang, Shiun-Zu Kuo, Hyokun Yun, Ejaz Ahmed, Shereen Oraby, Ziyun Li, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04120 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04120 diff --git a/papers/items/2026-2606-04296-the-saturation-trap-and-the-subjectivity-of-intervention-timing-why-affect-based.md b/papers/items/2026-2606-04296-the-saturation-trap-and-the-subjectivity-of-intervention-timing-why-affect-based.md new file mode 100644 index 0000000..b93c416 --- /dev/null +++ b/papers/items/2026-2606-04296-the-saturation-trap-and-the-subjectivity-of-intervention-timing-why-affect-based.md @@ -0,0 +1,63 @@ +# Paper: The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents + +--- +type: paper +title: "The Saturation Trap and the Subjectivity of Intervention Timing: Why Affect-Based Triggers and LLM Judges Fail to Time Interventions on Autonomous Agents" +authors: Manvendra Modgil +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04296 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-02 +updated_at: 2026-06-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, coding-agent, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04296 diff --git a/papers/items/2026-2606-04315-exploring-cross-scenario-generality-of-agentic-memory-systems-diagnostics-and-a-.md b/papers/items/2026-2606-04315-exploring-cross-scenario-generality-of-agentic-memory-systems-diagnostics-and-a-.md new file mode 100644 index 0000000..d4355c8 --- /dev/null +++ b/papers/items/2026-2606-04315-exploring-cross-scenario-generality-of-agentic-memory-systems-diagnostics-and-a-.md @@ -0,0 +1,62 @@ +# Paper: Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline + +--- +type: paper +title: "Exploring Cross-Scenario Generality of Agentic Memory Systems: Diagnostics and a Strong Baseline" +authors: Zhikai Chen, Jialiang Gu, Junyu Yin, Xianxuan Long, Shenglai Zeng, Xiaoze Liu, Kai Guo, Keren Zhou, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04315 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04315 diff --git a/papers/items/2026-2606-04555-temporal-order-matters-for-agentic-memory-segment-trees-for-long-horizon-agents.md b/papers/items/2026-2606-04555-temporal-order-matters-for-agentic-memory-segment-trees-for-long-horizon-agents.md new file mode 100644 index 0000000..2ea33af --- /dev/null +++ b/papers/items/2026-2606-04555-temporal-order-matters-for-agentic-memory-segment-trees-for-long-horizon-agents.md @@ -0,0 +1,62 @@ +# Paper: Temporal Order Matters for Agentic Memory: Segment Trees for Long-Horizon Agents + +--- +type: paper +title: "Temporal Order Matters for Agentic Memory: Segment Trees for Long-Horizon Agents" +authors: Yifan Simon Liu, Liam Gallagher, Faeze Moradi Kalarde, Jiazhou Liang, Armin Toroghi, Scott Sanner +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04555 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag +- arXiv categories: cs.CL, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04555 diff --git a/papers/items/2026-2606-04599-plan-first-judge-later-run-better-a-dmaic-inspired-agentic-system-for-industrial.md b/papers/items/2026-2606-04599-plan-first-judge-later-run-better-a-dmaic-inspired-agentic-system-for-industrial.md new file mode 100644 index 0000000..31d6b70 --- /dev/null +++ b/papers/items/2026-2606-04599-plan-first-judge-later-run-better-a-dmaic-inspired-agentic-system-for-industrial.md @@ -0,0 +1,63 @@ +# Paper: Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection + +--- +type: paper +title: "Plan First, Judge Later, Run Better: A DMAIC-Inspired Agentic System for Industrial Anomaly Detection" +authors: Yongzi Yu, Ao Li, Le Wang, Ziyue Li, Fugee Tsung, Yuxuan Liang, Man Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04599 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-safety + - multi-agent + - planning + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-safety, multi-agent, planning, rag, workflow-agent +- arXiv categories: cs.AI, cs.CE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04599 diff --git a/papers/items/2026-2606-04628-rampart-registry-based-agentic-memory-with-priority-aware-runtime-transformation.md b/papers/items/2026-2606-04628-rampart-registry-based-agentic-memory-with-priority-aware-runtime-transformation.md new file mode 100644 index 0000000..b6ba548 --- /dev/null +++ b/papers/items/2026-2606-04628-rampart-registry-based-agentic-memory-with-priority-aware-runtime-transformation.md @@ -0,0 +1,59 @@ +# Paper: RAMPART: Registry-based Agentic Memory with Priority-Aware Runtime Transformation + +--- +type: paper +title: "RAMPART: Registry-based Agentic Memory with Priority-Aware Runtime Transformation" +authors: Nikodem Tomczak +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04628 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: memory +- arXiv categories: cs.CL, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04628 diff --git a/papers/items/2026-2606-04780-personatree-structured-lifecycle-memory-for-person-understanding-in-llm-agents.md b/papers/items/2026-2606-04780-personatree-structured-lifecycle-memory-for-person-understanding-in-llm-agents.md new file mode 100644 index 0000000..4350e46 --- /dev/null +++ b/papers/items/2026-2606-04780-personatree-structured-lifecycle-memory-for-person-understanding-in-llm-agents.md @@ -0,0 +1,63 @@ +# Paper: PersonaTree: Structured Lifecycle Memory for Person Understanding in LLM Agents + +--- +type: paper +title: "PersonaTree: Structured Lifecycle Memory for Person Understanding in LLM Agents" +authors: Yubo Hou, Jingwei Song, Hongbo Zhang, Zhisheng Chen, Bang Xiao, Tao Wan, Zengchang Qin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04780 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, rag, tool-use +- arXiv categories: cs.CL +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04780 diff --git a/papers/items/2026-2606-04874-agent-planning-benchmark-a-diagnostic-framework-for-planning-capabilities-in-llm.md b/papers/items/2026-2606-04874-agent-planning-benchmark-a-diagnostic-framework-for-planning-capabilities-in-llm.md new file mode 100644 index 0000000..34ef3f4 --- /dev/null +++ b/papers/items/2026-2606-04874-agent-planning-benchmark-a-diagnostic-framework-for-planning-capabilities-in-llm.md @@ -0,0 +1,61 @@ +# Paper: Agent Planning Benchmark: A Diagnostic Framework for Planning Capabilities in LLM Agents + +--- +type: paper +title: "Agent Planning Benchmark: A Diagnostic Framework for Planning Capabilities in LLM Agents" +authors: Haoyu Sun, Wenxuan Wang, Mingyang Song, Jujie He, Weinan Zhang, Yang Liu, Yang Yang, Yu Cheng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04874 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-05 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, planning-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use +- arXiv categories: cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04874 diff --git a/papers/items/2026-2606-04990-from-agent-traces-to-trust-a-survey-of-evidence-tracing-and-execution-provenance.md b/papers/items/2026-2606-04990-from-agent-traces-to-trust-a-survey-of-evidence-tracing-and-execution-provenance.md new file mode 100644 index 0000000..650e480 --- /dev/null +++ b/papers/items/2026-2606-04990-from-agent-traces-to-trust-a-survey-of-evidence-tracing-and-execution-provenance.md @@ -0,0 +1,65 @@ +# Paper: From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents + +--- +type: paper +title: "From Agent Traces to Trust: A Survey of Evidence Tracing and Execution Provenance in LLM Agents" +authors: Yiqi Wang, Jiaqi Zhang, Taotao Cai, Zirui Liu, Qingqiang Sun, Zequn Sun, Zhangkai Wu, Manqing Dong, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.04990 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - multi-agent + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, memory, multi-agent, planning, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.04990 diff --git a/papers/items/2026-2606-05241-search-time-contamination-in-deep-research-agents-measuring-performance-inflatio.md b/papers/items/2026-2606-05241-search-time-contamination-in-deep-research-agents-measuring-performance-inflatio.md new file mode 100644 index 0000000..8222b99 --- /dev/null +++ b/papers/items/2026-2606-05241-search-time-contamination-in-deep-research-agents-measuring-performance-inflatio.md @@ -0,0 +1,61 @@ +# Paper: Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation + +--- +type: paper +title: "Search-Time Contamination in Deep Research Agents: Measuring Performance Inflation in Public Benchmark Evaluation" +authors: Yongjie Wang, Xinyue Zhang, Kunhong Yao, Zhiwei Zeng, Kaisong Song, Jun Lin, Zhiqi Shen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05241 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, rag, reasoning +- arXiv categories: cs.CR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05241 diff --git a/papers/items/2026-2606-05263-policy-conditioned-counterfactual-credit-for-verifiable-reinforcement-learning-o.md b/papers/items/2026-2606-05263-policy-conditioned-counterfactual-credit-for-verifiable-reinforcement-learning-o.md new file mode 100644 index 0000000..4a4b99e --- /dev/null +++ b/papers/items/2026-2606-05263-policy-conditioned-counterfactual-credit-for-verifiable-reinforcement-learning-o.md @@ -0,0 +1,63 @@ +# Paper: Policy-Conditioned Counterfactual Credit for Verifiable Reinforcement Learning of Long-Horizon Language Agents + +--- +type: paper +title: Policy-Conditioned Counterfactual Credit for Verifiable Reinforcement Learning of Long-Horizon Language Agents +authors: Renwei Meng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05263 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use +- arXiv categories: cs.LG, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05263 diff --git a/papers/items/2026-2606-05414-when-evidence-is-sparse-weakly-supervised-early-failure-alerting-in-dialogs-and-.md b/papers/items/2026-2606-05414-when-evidence-is-sparse-weakly-supervised-early-failure-alerting-in-dialogs-and-.md new file mode 100644 index 0000000..fe1795f --- /dev/null +++ b/papers/items/2026-2606-05414-when-evidence-is-sparse-weakly-supervised-early-failure-alerting-in-dialogs-and-.md @@ -0,0 +1,65 @@ +# Paper: When Evidence is Sparse: Weakly Supervised Early Failure Alerting in Dialogs and LLM-Agent Trajectories + +--- +type: paper +title: "When Evidence is Sparse: Weakly Supervised Early Failure Alerting in Dialogs and LLM-Agent Trajectories" +authors: Avinash Baidya, Xinran Liang, Ruocheng Guo, Xiang Gao, Kamalika Das +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05414 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.HC + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, rag, tool-use +- arXiv categories: cs.CL, cs.AI, cs.HC, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05414 diff --git a/papers/items/2026-2606-05436-ten-headache-specialists-versus-artificial-intelligence-for-clinical-literature-.md b/papers/items/2026-2606-05436-ten-headache-specialists-versus-artificial-intelligence-for-clinical-literature-.md new file mode 100644 index 0000000..4fa4b5f --- /dev/null +++ b/papers/items/2026-2606-05436-ten-headache-specialists-versus-artificial-intelligence-for-clinical-literature-.md @@ -0,0 +1,63 @@ +# Paper: Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison + +--- +type: paper +title: "Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison" +authors: Alejandro Lozano, Keiko Ihara, Ping-Hao Yang, Carrie E. Robertson, Jennifer Stern, Allan Purdy, Hsiangkuo Yuan, Pengfei Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05436 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, computer-use, rag, tool-use +- arXiv categories: cs.AI, cs.CL, cs.IR +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05436 diff --git a/papers/items/2026-2606-05463-psebench-a-controllable-and-verifiable-benchmark-for-evaluating-llms-in-patient-.md b/papers/items/2026-2606-05463-psebench-a-controllable-and-verifiable-benchmark-for-evaluating-llms-in-patient-.md new file mode 100644 index 0000000..f9667f9 --- /dev/null +++ b/papers/items/2026-2606-05463-psebench-a-controllable-and-verifiable-benchmark-for-evaluating-llms-in-patient-.md @@ -0,0 +1,62 @@ +# Paper: PSEBench: A Controllable and Verifiable Benchmark for Evaluating LLMs in Patient Safety Event Triage + +--- +type: paper +title: "PSEBench: A Controllable and Verifiable Benchmark for Evaluating LLMs in Patient Safety Event Triage" +authors: Keqi Han, Ryan Young, Annabel Strauss, Lindsey Hughes, Katharine M. Nesbitt, Nicole Schueler, Che Ngufor, Carl Yang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05463 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05463 diff --git a/papers/items/2026-2606-05548-adk-arena-evaluating-agent-development-kits-via-llm-as-a-developer.md b/papers/items/2026-2606-05548-adk-arena-evaluating-agent-development-kits-via-llm-as-a-developer.md new file mode 100644 index 0000000..4235ebb --- /dev/null +++ b/papers/items/2026-2606-05548-adk-arena-evaluating-agent-development-kits-via-llm-as-a-developer.md @@ -0,0 +1,61 @@ +# Paper: ADK Arena: Evaluating Agent Development Kits via LLM-as-a-Developer + +--- +type: paper +title: "ADK Arena: Evaluating Agent Development Kits via LLM-as-a-Developer" +authors: Jintao Huang, Xiaomin Li, Gaurav Mittal, Yu Hu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05548 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation, autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, autonomous-agent-llm +- inferred topics: agent-evaluation, coding-agent, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05548 diff --git a/papers/items/2026-2606-05558-autoregressive-diffusion-world-models-for-off-policy-evaluation-of-llm-agents.md b/papers/items/2026-2606-05558-autoregressive-diffusion-world-models-for-off-policy-evaluation-of-llm-agents.md new file mode 100644 index 0000000..e524eff --- /dev/null +++ b/papers/items/2026-2606-05558-autoregressive-diffusion-world-models-for-off-policy-evaluation-of-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: Autoregressive Diffusion World Models for Off-Policy Evaluation of LLM Agents + +--- +type: paper +title: Autoregressive Diffusion World Models for Off-Policy Evaluation of LLM Agents +authors: Kaixuan Liu, Guojun Xiong, Weinan Zhang, Shengpu Tang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05558 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, computer-use, tool-use, world-model +- arXiv categories: cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05558 diff --git a/papers/items/2026-2606-05622-adaplanbench-evaluating-adaptive-planning-in-large-language-model-agents-under-w.md b/papers/items/2026-2606-05622-adaplanbench-evaluating-adaptive-planning-in-large-language-model-agents-under-w.md new file mode 100644 index 0000000..be1e410 --- /dev/null +++ b/papers/items/2026-2606-05622-adaplanbench-evaluating-adaptive-planning-in-large-language-model-agents-under-w.md @@ -0,0 +1,61 @@ +# Paper: AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints + +--- +type: paper +title: "AdaPlanBench: Evaluating Adaptive Planning in Large Language Model Agents under World and User Constraints" +authors: Jiayu Liu, Cheng Qian, Zhenhailong Wang, Bingxuan Li, Jiateng Liu, Heng Wang, Jeonghwan Kim, Yumeng Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05622 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, tool-use +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05622 diff --git a/papers/items/2026-2606-05658-agent-orchestrated-adaptive-rag-a-comparative-study-on-structured-and-multi-hop-.md b/papers/items/2026-2606-05658-agent-orchestrated-adaptive-rag-a-comparative-study-on-structured-and-multi-hop-.md new file mode 100644 index 0000000..bee8723 --- /dev/null +++ b/papers/items/2026-2606-05658-agent-orchestrated-adaptive-rag-a-comparative-study-on-structured-and-multi-hop-.md @@ -0,0 +1,61 @@ +# Paper: Agent-Orchestrated Adaptive RAG: A Comparative Study on Structured and Multi-Hop Retrieval + +--- +type: paper +title: "Agent-Orchestrated Adaptive RAG: A Comparative Study on Structured and Multi-Hop Retrieval" +authors: Anuj Maharjan, Devinder Kaur, Richard Molyet +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05658 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, rag, reasoning +- arXiv categories: cs.IR, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05658 diff --git a/papers/items/2026-2606-05684-adamem-test-time-adaptive-memory-for-language-agents.md b/papers/items/2026-2606-05684-adamem-test-time-adaptive-memory-for-language-agents.md new file mode 100644 index 0000000..99642ad --- /dev/null +++ b/papers/items/2026-2606-05684-adamem-test-time-adaptive-memory-for-language-agents.md @@ -0,0 +1,63 @@ +# Paper: AdaMEM: Test-Time Adaptive Memory for Language Agents + +--- +type: paper +title: "AdaMEM: Test-Time Adaptive Memory for Language Agents" +authors: Yunxiang Zhang, Yiheng Li, Ali Payani, Lu Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05684 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory, language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, language-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, rag, reasoning +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05684 diff --git a/papers/items/2026-2606-05711-beyond-tokens-a-unified-framework-for-latent-communication-in-llm-based-multi-ag.md b/papers/items/2026-2606-05711-beyond-tokens-a-unified-framework-for-latent-communication-in-llm-based-multi-ag.md new file mode 100644 index 0000000..ccf7307 --- /dev/null +++ b/papers/items/2026-2606-05711-beyond-tokens-a-unified-framework-for-latent-communication-in-llm-based-multi-ag.md @@ -0,0 +1,63 @@ +# Paper: Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems + +--- +type: paper +title: "Beyond tokens: a unified framework for latent communication in LLM-based multi-agent systems" +authors: Yingzhuo Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05711 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-05 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-safety, computer-use, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05711 diff --git a/papers/items/2026-2606-05805-from-risk-classification-to-action-plan-remediation-a-guardrail-feedback-driven-.md b/papers/items/2026-2606-05805-from-risk-classification-to-action-plan-remediation-a-guardrail-feedback-driven-.md new file mode 100644 index 0000000..d0299a4 --- /dev/null +++ b/papers/items/2026-2606-05805-from-risk-classification-to-action-plan-remediation-a-guardrail-feedback-driven-.md @@ -0,0 +1,63 @@ +# Paper: From Risk Classification to Action Plan Remediation: A Guardrail Feedback Driven Framework for LLM Agents + +--- +type: paper +title: "From Risk Classification to Action Plan Remediation: A Guardrail Feedback Driven Framework for LLM Agents" +authors: Yuhao Sun, Jiacheng Zhang, Shaanan Cohney, Zhexin Zhang, Feng Liu, Xingliang Yuan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.05805 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.05805 diff --git a/papers/items/2026-2606-06054-beyond-similarity-trustworthy-memory-search-for-personal-ai-agents.md b/papers/items/2026-2606-06054-beyond-similarity-trustworthy-memory-search-for-personal-ai-agents.md new file mode 100644 index 0000000..2f5a271 --- /dev/null +++ b/papers/items/2026-2606-06054-beyond-similarity-trustworthy-memory-search-for-personal-ai-agents.md @@ -0,0 +1,60 @@ +# Paper: Beyond Similarity: Trustworthy Memory Search for Personal AI Agents + +--- +type: paper +title: "Beyond Similarity: Trustworthy Memory Search for Personal AI Agents" +authors: Jiawen Zhang, Kejia Chen, Jiachen Ma, Yangfan Hu, Lipeng He, Yechao Zhang, Jian Liu, Xiaohu Yang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.06054 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.06054 diff --git a/papers/items/2026-2606-06090-beyond-semantic-organization-memory-as-execution-state-management-for-long-horiz.md b/papers/items/2026-2606-06090-beyond-semantic-organization-memory-as-execution-state-management-for-long-horiz.md new file mode 100644 index 0000000..386cc93 --- /dev/null +++ b/papers/items/2026-2606-06090-beyond-semantic-organization-memory-as-execution-state-management-for-long-horiz.md @@ -0,0 +1,62 @@ +# Paper: Beyond Semantic Organization: Memory as Execution State Management for Long-Horizon Agents + +--- +type: paper +title: "Beyond Semantic Organization: Memory as Execution State Management for Long-Horizon Agents" +authors: Yaoqi Chen, Haibin Lai, Yuru Feng, Chuyu Han, Qianxi Zhang, Baotong Lu, Menghao Li, Xinjiang Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.06090 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - computer-use + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, rag-agent +- inferred topics: computer-use, memory, planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.06090 diff --git a/papers/items/2026-2606-06388-humans-almanac-a-human-collaboration-dataset-of-action-level-mental-model-annota.md b/papers/items/2026-2606-06388-humans-almanac-a-human-collaboration-dataset-of-action-level-mental-model-annota.md new file mode 100644 index 0000000..a2852f7 --- /dev/null +++ b/papers/items/2026-2606-06388-humans-almanac-a-human-collaboration-dataset-of-action-level-mental-model-annota.md @@ -0,0 +1,64 @@ +# Paper: Humans' ALMANAC: A Human Collaboration Dataset of Action-Level Mental Model Annotations for Agent Collaboration + +--- +type: paper +title: "Humans' ALMANAC: A Human Collaboration Dataset of Action-Level Mental Model Annotations for Agent Collaboration" +authors: Jiaju Chen, Yuxuan Lu, Jiayi Su, Chaoran Chen, Songlin Xiao, Zheng Zhang, Yun Wang, Yunyao Li, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.06388 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-06 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.06388 diff --git a/papers/items/2026-2606-06399-collabsim-a-cscw-grounded-methodology-for-investigating-collaborative-competence.md b/papers/items/2026-2606-06399-collabsim-a-cscw-grounded-methodology-for-investigating-collaborative-competence.md new file mode 100644 index 0000000..1865988 --- /dev/null +++ b/papers/items/2026-2606-06399-collabsim-a-cscw-grounded-methodology-for-investigating-collaborative-competence.md @@ -0,0 +1,64 @@ +# Paper: CollabSim: A CSCW-Grounded Methodology for Investigating Collaborative Competence of LLM Agents through Controlled Multi-Agent Experiments + +--- +type: paper +title: "CollabSim: A CSCW-Grounded Methodology for Investigating Collaborative Competence of LLM Agents through Controlled Multi-Agent Experiments" +authors: Jiaju Chen, Bo Sun, Yuxuan Lu, Yun Wang, Dakuo Wang, Bingsheng Yao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.06399 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - planning + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, multi-agent, planning, reasoning, tool-use, world-model +- arXiv categories: cs.CL +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.06399 diff --git a/papers/items/2026-2606-06448-agent-memory-characterization-and-system-implications-of-stateful-long-horizon-w.md b/papers/items/2026-2606-06448-agent-memory-characterization-and-system-implications-of-stateful-long-horizon-w.md new file mode 100644 index 0000000..cb0ac4c --- /dev/null +++ b/papers/items/2026-2606-06448-agent-memory-characterization-and-system-implications-of-stateful-long-horizon-w.md @@ -0,0 +1,63 @@ +# Paper: Agent Memory: Characterization and System Implications of Stateful Long-Horizon Workloads + +--- +type: paper +title: "Agent Memory: Characterization and System Implications of Stateful Long-Horizon Workloads" +authors: Yasmine Omri, Ziyu Gan, Zachary Broveak, Robin Geens, Zexue He, Alex Pentland, Marian Verhelst, Tsachy Weissman, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.06448 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.06448 diff --git a/papers/items/2026-2606-06462-benchmark-everything-everywhere-all-at-once.md b/papers/items/2026-2606-06462-benchmark-everything-everywhere-all-at-once.md new file mode 100644 index 0000000..e9f45c2 --- /dev/null +++ b/papers/items/2026-2606-06462-benchmark-everything-everywhere-all-at-once.md @@ -0,0 +1,61 @@ +# Paper: Benchmark Everything Everywhere All at Once + +--- +type: paper +title: Benchmark Everything Everywhere All at Once +authors: Shiyun Xiong, Dongming Wu, Peiwen Sun, Yuang Ai, Bokang Yang, Wencheng Han, Xiao-Hui Li, Xiangyu Yue +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.06462 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, coding-agent, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.06462 diff --git a/papers/items/2026-2606-06473-mlevolve-a-self-evolving-framework-for-automated-machine-learning-algorithm-disc.md b/papers/items/2026-2606-06473-mlevolve-a-self-evolving-framework-for-automated-machine-learning-algorithm-disc.md new file mode 100644 index 0000000..4ce7160 --- /dev/null +++ b/papers/items/2026-2606-06473-mlevolve-a-self-evolving-framework-for-automated-machine-learning-algorithm-disc.md @@ -0,0 +1,64 @@ +# Paper: MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery + +--- +type: paper +title: "MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery" +authors: Shangheng Du, Xiangchao Yan, Jinxin Shi, Zongsheng Cao, Shiyang Feng, Zichen Liang, Boyuan Sun, Tianshuo Peng, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.06473 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-04 +updated_at: 2026-06-04 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - multi-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, coding-agent, memory, multi-agent, planning, rag +- arXiv categories: cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.06473 diff --git a/papers/items/2026-2606-07314-qbuglm-an-agentic-benchmarking-framework-for-llm-based-quantum-software-debuggin.md b/papers/items/2026-2606-07314-qbuglm-an-agentic-benchmarking-framework-for-llm-based-quantum-software-debuggin.md new file mode 100644 index 0000000..defc7de --- /dev/null +++ b/papers/items/2026-2606-07314-qbuglm-an-agentic-benchmarking-framework-for-llm-based-quantum-software-debuggin.md @@ -0,0 +1,64 @@ +# Paper: QBugLM: An Agentic Benchmarking Framework for LLM-based Quantum Software Debugging + +--- +type: paper +title: "QBugLM: An Agentic Benchmarking Framework for LLM-based Quantum Software Debugging" +authors: An B. B. Pham, Hoa T. Nguyen, Muhammad Usman +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.07314 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-05 +updated_at: 2026-06-05 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - reasoning + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.ET + - quant-ph +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, multi-agent, reasoning, world-model +- arXiv categories: cs.SE, cs.ET, quant-ph +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.07314 diff --git a/papers/items/2026-2606-07379-do-coding-agents-deceive-us-detecting-and-preventing-cheating-via-capped-evaluat.md b/papers/items/2026-2606-07379-do-coding-agents-deceive-us-detecting-and-preventing-cheating-via-capped-evaluat.md new file mode 100644 index 0000000..eef31c5 --- /dev/null +++ b/papers/items/2026-2606-07379-do-coding-agents-deceive-us-detecting-and-preventing-cheating-via-capped-evaluat.md @@ -0,0 +1,63 @@ +# Paper: Do Coding Agents Deceive Us? Detecting and Preventing Cheating via Capped Evaluation with Randomized Tests + +--- +type: paper +title: Do Coding Agents Deceive Us? Detecting and Preventing Cheating via Capped Evaluation with Randomized Tests +authors: Thanawat Lodkaew, Johannes Ackermann, Soichiro Nishimori, Nontawat Charoenphakdee, Masashi Sugiyama, Takashi Ishida +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.07379 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-05 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL + - stat.ME +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, rag +- arXiv categories: cs.LG, cs.AI, cs.CL, stat.ME +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.07379 diff --git a/papers/items/2026-2606-07402-m-3-exam-benchmarking-multimodal-memory-for-realistic-user-agent-interactions.md b/papers/items/2026-2606-07402-m-3-exam-benchmarking-multimodal-memory-for-realistic-user-agent-interactions.md new file mode 100644 index 0000000..cb63ea4 --- /dev/null +++ b/papers/items/2026-2606-07402-m-3-exam-benchmarking-multimodal-memory-for-realistic-user-agent-interactions.md @@ -0,0 +1,62 @@ +# Paper: M$^3$Exam: Benchmarking Multimodal Memory for Realistic User-Agent Interactions + +--- +type: paper +title: "M$^3$Exam: Benchmarking Multimodal Memory for Realistic User-Agent Interactions" +authors: Zhengjun Huang, Wenxuan Liu, Zhoujin Tian, Wei Chen, Junle Chen, Yuqian Wu, Fangyuan Zhang, Qintian Guo, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.07402 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-05 +updated_at: 2026-06-05 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.07402 diff --git a/papers/items/2026-2606-07591-researchclawbench-a-benchmark-for-end-to-end-autonomous-scientific-research.md b/papers/items/2026-2606-07591-researchclawbench-a-benchmark-for-end-to-end-autonomous-scientific-research.md new file mode 100644 index 0000000..8d44ed4 --- /dev/null +++ b/papers/items/2026-2606-07591-researchclawbench-a-benchmark-for-end-to-end-autonomous-scientific-research.md @@ -0,0 +1,62 @@ +# Paper: ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research + +--- +type: paper +title: "ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research" +authors: Wanghan Xu, Shuo Li, Tianlin Ye, Qinglong Cao, Yixin Chen, Hengjian Gao, Yiheng Wang, Qi Li, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.07591 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, coding-agent, rag +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.07591 diff --git a/papers/items/2026-2606-07595-visualleakbench-reproducible-action-boundary-propagation-failures-in-vision-lang.md b/papers/items/2026-2606-07595-visualleakbench-reproducible-action-boundary-propagation-failures-in-vision-lang.md new file mode 100644 index 0000000..e34b6e6 --- /dev/null +++ b/papers/items/2026-2606-07595-visualleakbench-reproducible-action-boundary-propagation-failures-in-vision-lang.md @@ -0,0 +1,64 @@ +# Paper: VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents + +--- +type: paper +title: "VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents" +authors: Youting Wang, Yuan Tang, Yitian Qian, Chen Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.07595 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-29 +updated_at: 2026-05-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, agent-safety, memory, tool-use, workflow-agent +- arXiv categories: cs.CV, cs.AI, cs.IR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.07595 diff --git a/papers/items/2026-2606-07682-swe-marathon-can-agents-autonomously-complete-ultra-long-horizon-software-work.md b/papers/items/2026-2606-07682-swe-marathon-can-agents-autonomously-complete-ultra-long-horizon-software-work.md new file mode 100644 index 0000000..703996f --- /dev/null +++ b/papers/items/2026-2606-07682-swe-marathon-can-agents-autonomously-complete-ultra-long-horizon-software-work.md @@ -0,0 +1,65 @@ +# Paper: SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work? + +--- +type: paper +title: "SWE-Marathon: Can Agents Autonomously Complete Ultra-Long-Horizon Software Work?" +authors: Rishi Desai, Jesse Hu, Joan Cabezas, Neel Harsola, Pratyush Shukla, Roey Ben Chaim, Adnan El Assadi, Omkaar Mukund Kamath, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.07682 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-05 +updated_at: 2026-06-05 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - planning + - rag + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, memory, planning, rag, reasoning, workflow-agent +- arXiv categories: cs.SE, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.07682 diff --git a/papers/items/2026-2606-07711-rosetta-memory-adaptive-memory-for-cross-llm-agents.md b/papers/items/2026-2606-07711-rosetta-memory-adaptive-memory-for-cross-llm-agents.md new file mode 100644 index 0000000..d2c86e8 --- /dev/null +++ b/papers/items/2026-2606-07711-rosetta-memory-adaptive-memory-for-cross-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: Rosetta Memory: Adaptive Memory for Cross-LLM Agents + +--- +type: paper +title: "Rosetta Memory: Adaptive Memory for Cross-LLM Agents" +authors: Hao Yang, Shiqi Shen, Haoxuan Li, Zhipeng Wang, Zhi Gong, Xu Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.07711 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-05 +updated_at: 2026-06-05 +status: queued +relevance: high +topics: + - coding-agent + - memory + - planning + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: coding-agent, memory, planning, reasoning +- arXiv categories: cs.LG, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.07711 diff --git a/papers/items/2026-2606-07836-agentic-multi-fidelity-learning-of-quasiparticle-and-excitonic-properties.md b/papers/items/2026-2606-07836-agentic-multi-fidelity-learning-of-quasiparticle-and-excitonic-properties.md new file mode 100644 index 0000000..e2af279 --- /dev/null +++ b/papers/items/2026-2606-07836-agentic-multi-fidelity-learning-of-quasiparticle-and-excitonic-properties.md @@ -0,0 +1,66 @@ +# Paper: Agentic multi-fidelity learning of quasiparticle and excitonic properties + +--- +type: paper +title: Agentic multi-fidelity learning of quasiparticle and excitonic properties +authors: Arnab Neogi, Aaron Forde, Christopher A. Lane, Sergei Tretiak, Jian-Xin Zhu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.07836 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-05 +updated_at: 2026-06-05 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cond-mat.mtrl-sci + - cond-mat.stat-mech + - cs.AI + - physics.comp-ph + - quant-ph +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, computer-use, rag, workflow-agent, world-model +- arXiv categories: cond-mat.mtrl-sci, cond-mat.stat-mech, cs.AI, physics.comp-ph, quant-ph +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.07836 diff --git a/papers/items/2026-2606-07867-the-cold-start-safety-gap-in-llm-agents.md b/papers/items/2026-2606-07867-the-cold-start-safety-gap-in-llm-agents.md new file mode 100644 index 0000000..8d65402 --- /dev/null +++ b/papers/items/2026-2606-07867-the-cold-start-safety-gap-in-llm-agents.md @@ -0,0 +1,60 @@ +# Paper: The Cold-Start Safety Gap in LLM Agents + +--- +type: paper +title: The Cold-Start Safety Gap in LLM Agents +authors: Chung-En Sun, Linbo Liu, Tsui-Wei Weng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.07867 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-05 +updated_at: 2026-06-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.07867 diff --git a/papers/items/2026-2606-08162-silent-failure-in-llm-agent-systems-the-entropy-principle-and-the-inevitable-dis.md b/papers/items/2026-2606-08162-silent-failure-in-llm-agent-systems-the-entropy-principle-and-the-inevitable-dis.md new file mode 100644 index 0000000..ff77c21 --- /dev/null +++ b/papers/items/2026-2606-08162-silent-failure-in-llm-agent-systems-the-entropy-principle-and-the-inevitable-dis.md @@ -0,0 +1,59 @@ +# Paper: Silent Failure in LLM Agent Systems: The Entropy Principle and the Inevitable Disorder of Autonomous Agents + +--- +type: paper +title: "Silent Failure in LLM Agent Systems: The Entropy Principle and the Inevitable Disorder of Autonomous Agents" +authors: Dexing Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.08162 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-06 +updated_at: 2026-06-06 +status: queued +relevance: high +topics: + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: memory, tool-use +- arXiv categories: cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.08162 diff --git a/papers/items/2026-2606-08172-the-governance-of-human-llm-interaction-safety-gating-civility-steering-and-affe.md b/papers/items/2026-2606-08172-the-governance-of-human-llm-interaction-safety-gating-civility-steering-and-affe.md new file mode 100644 index 0000000..4745151 --- /dev/null +++ b/papers/items/2026-2606-08172-the-governance-of-human-llm-interaction-safety-gating-civility-steering-and-affe.md @@ -0,0 +1,65 @@ +# Paper: The Governance of Human-LLM Interaction: Safety Gating, Civility Steering, and Affective Default Lock-In + +--- +type: paper +title: "The Governance of Human-LLM Interaction: Safety Gating, Civility Steering, and Affective Default Lock-In" +authors: Manuele Reani, Hongjian Zhang, Hongyu Tian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.08172 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-06 +updated_at: 2026-06-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.HC + - cs.AI + - cs.CY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, planning, tool-use +- arXiv categories: cs.HC, cs.AI, cs.CY +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.08172 diff --git a/papers/items/2026-2606-08274-toward-human-centered-multi-agent-systems-integrating-cognition-culture-values-a.md b/papers/items/2026-2606-08274-toward-human-centered-multi-agent-systems-integrating-cognition-culture-values-a.md new file mode 100644 index 0000000..9e31382 --- /dev/null +++ b/papers/items/2026-2606-08274-toward-human-centered-multi-agent-systems-integrating-cognition-culture-values-a.md @@ -0,0 +1,64 @@ +# Paper: Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents + +--- +type: paper +title: "Toward Human-Centered Multi-Agent Systems: Integrating Cognition, Culture, Values, and Cooperation in AI Agents" +authors: Safia Baloch, Rahemeen Khan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.08274 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-06 +updated_at: 2026-06-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 24 +collection_queries: autonomous-agent-llm, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, planning-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.MA +- collection score: 24 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.08274 diff --git a/papers/items/2026-2606-08340-benchmarking-open-ended-multi-agent-coordination-in-language-agents.md b/papers/items/2026-2606-08340-benchmarking-open-ended-multi-agent-coordination-in-language-agents.md new file mode 100644 index 0000000..6301bc4 --- /dev/null +++ b/papers/items/2026-2606-08340-benchmarking-open-ended-multi-agent-coordination-in-language-agents.md @@ -0,0 +1,66 @@ +# Paper: Benchmarking Open-Ended Multi-Agent Coordination in Language Agents + +--- +type: paper +title: Benchmarking Open-Ended Multi-Agent Coordination in Language Agents +authors: Kale-ab Abebe Tessera, Andras Szecsenyi, Cameron Barker, Alexander Rutherford, Davide Paglieri, Aidan Scannell, Henry Gouk, Elliot J. Crowley, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.08340 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-06 +updated_at: 2026-06-06 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 28 +collection_queries: autonomous-agent-llm, language-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, language-agent, planning-agent +- inferred topics: agent-evaluation, memory, multi-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.LG, cs.MA +- collection score: 28 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.08340 diff --git a/papers/items/2026-2606-08531-vesta-a-fully-automated-scenario-generation-and-safety-evaluation-framework-for-.md b/papers/items/2026-2606-08531-vesta-a-fully-automated-scenario-generation-and-safety-evaluation-framework-for-.md new file mode 100644 index 0000000..0dcd981 --- /dev/null +++ b/papers/items/2026-2606-08531-vesta-a-fully-automated-scenario-generation-and-safety-evaluation-framework-for-.md @@ -0,0 +1,62 @@ +# Paper: VESTA: A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents + +--- +type: paper +title: "VESTA: A Fully Automated Scenario Generation and Safety Evaluation Framework for LLM Agents" +authors: Lu Jia, Haibo Tong, Feifei Zhao, Jindong Li, Dongqi Liang, Ping Wu, Qian Zhang, Yi Zeng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.08531 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-07 +updated_at: 2026-06-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.08531 diff --git a/papers/items/2026-2606-08625-from-holistic-evaluation-to-structured-criteria-rubrics-across-the-evolving-llm-.md b/papers/items/2026-2606-08625-from-holistic-evaluation-to-structured-criteria-rubrics-across-the-evolving-llm-.md new file mode 100644 index 0000000..685bb68 --- /dev/null +++ b/papers/items/2026-2606-08625-from-holistic-evaluation-to-structured-criteria-rubrics-across-the-evolving-llm-.md @@ -0,0 +1,62 @@ +# Paper: From Holistic Evaluation to Structured Criteria: Rubrics Across the Evolving LLM Landscape + +--- +type: paper +title: "From Holistic Evaluation to Structured Criteria: Rubrics Across the Evolving LLM Landscape" +authors: Hao Chen, Ziyu Han, Yukun Yan, Qingfu Zhu, Maosong Sun, Wanxiang Che +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.08625 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-07 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, computer-use, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.08625 diff --git a/papers/items/2026-2606-08790-rails-verification-native-clearing-for-agentic-commerce.md b/papers/items/2026-2606-08790-rails-verification-native-clearing-for-agentic-commerce.md new file mode 100644 index 0000000..f854cd8 --- /dev/null +++ b/papers/items/2026-2606-08790-rails-verification-native-clearing-for-agentic-commerce.md @@ -0,0 +1,62 @@ +# Paper: RAILS: Verification-Native Clearing For Agentic Commerce + +--- +type: paper +title: "RAILS: Verification-Native Clearing For Agentic Commerce" +authors: Adrian de Valois-Franklin, Alex Bogdan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.08790 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-07 +updated_at: 2026-06-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CR + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.AI, cs.CR, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.08790 diff --git a/papers/items/2026-2606-08960-hardening-agent-benchmarks-with-adversarial-hacker-fixer-loops.md b/papers/items/2026-2606-08960-hardening-agent-benchmarks-with-adversarial-hacker-fixer-loops.md new file mode 100644 index 0000000..115a9da --- /dev/null +++ b/papers/items/2026-2606-08960-hardening-agent-benchmarks-with-adversarial-hacker-fixer-loops.md @@ -0,0 +1,62 @@ +# Paper: Hardening Agent Benchmarks with Adversarial Hacker-Fixer Loops + +--- +type: paper +title: Hardening Agent Benchmarks with Adversarial Hacker-Fixer Loops +authors: Ziqian Zhong, Ivgeni Segal, Ivan Bercovich, Shashwat Saxena, Kexun Zhang, Aditi Raghunathan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.08960 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, reasoning +- arXiv categories: cs.CR, cs.AI, cs.LG, cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.08960 diff --git a/papers/items/2026-2606-09037-a-multi-agent-system-for-ipmsm-design-optimization-via-an-fea-ai-hybrid-approach.md b/papers/items/2026-2606-09037-a-multi-agent-system-for-ipmsm-design-optimization-via-an-fea-ai-hybrid-approach.md new file mode 100644 index 0000000..bfb53bd --- /dev/null +++ b/papers/items/2026-2606-09037-a-multi-agent-system-for-ipmsm-design-optimization-via-an-fea-ai-hybrid-approach.md @@ -0,0 +1,64 @@ +# Paper: A Multi-Agent System for IPMSM Design Optimization via an FEA-AI Hybrid Approach + +--- +type: paper +title: A Multi-Agent System for IPMSM Design Optimization via an FEA-AI Hybrid Approach +authors: Jinseong Han, Sunwoong Yang, Namwoo Kang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09037 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - rag + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, multi-agent, planning, rag, reasoning, workflow-agent +- arXiv categories: cs.AI, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09037 diff --git a/papers/items/2026-2606-09071-reflect-intervention-supported-error-attribution-for-silent-failures-in-llm-agen.md b/papers/items/2026-2606-09071-reflect-intervention-supported-error-attribution-for-silent-failures-in-llm-agen.md new file mode 100644 index 0000000..bc0f4d1 --- /dev/null +++ b/papers/items/2026-2606-09071-reflect-intervention-supported-error-attribution-for-silent-failures-in-llm-agen.md @@ -0,0 +1,61 @@ +# Paper: REFLECT: Intervention-Supported Error Attribution for Silent Failures in LLM Agent Traces + +--- +type: paper +title: "REFLECT: Intervention-Supported Error Attribution for Silent Failures in LLM Agent Traces" +authors: Xiaofeng Lin, Yingxu Wang, Tung Sum Thomas Kwok, Daniel Guo, Sahil Arun Nale, Charles Fleming, Guang Cheng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09071 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09071 diff --git a/papers/items/2026-2606-09198-mass-deep-research-for-social-sciences-with-memory-augmented-social-simulation.md b/papers/items/2026-2606-09198-mass-deep-research-for-social-sciences-with-memory-augmented-social-simulation.md new file mode 100644 index 0000000..16997d7 --- /dev/null +++ b/papers/items/2026-2606-09198-mass-deep-research-for-social-sciences-with-memory-augmented-social-simulation.md @@ -0,0 +1,63 @@ +# Paper: MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation + +--- +type: paper +title: "MASS: Deep Research for Social Sciences with Memory-Augmented Social Simulation" +authors: Yongrui Liu, Deyi Xiong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09198 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, planning, rag, world-model +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09198 diff --git a/papers/items/2026-2606-09316-anything2skill-compiling-external-knowledge-into-reusable-skills-for-agents.md b/papers/items/2026-2606-09316-anything2skill-compiling-external-knowledge-into-reusable-skills-for-agents.md new file mode 100644 index 0000000..d4f7a73 --- /dev/null +++ b/papers/items/2026-2606-09316-anything2skill-compiling-external-knowledge-into-reusable-skills-for-agents.md @@ -0,0 +1,64 @@ +# Paper: Anything2Skill: Compiling External Knowledge into Reusable Skills for Agents + +--- +type: paper +title: "Anything2Skill: Compiling External Knowledge into Reusable Skills for Agents" +authors: Qianjun Pan, Yutao Yang, Junsong Li, Jie Zhou, Kai Chen, Xin Li, Qin Chen, Liang He +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09316 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09316 diff --git a/papers/items/2026-2606-09399-runagent-superbrowser-a-theory-of-autonomous-web-navigation-grounded-in-human-br.md b/papers/items/2026-2606-09399-runagent-superbrowser-a-theory-of-autonomous-web-navigation-grounded-in-human-br.md new file mode 100644 index 0000000..917d32a --- /dev/null +++ b/papers/items/2026-2606-09399-runagent-superbrowser-a-theory-of-autonomous-web-navigation-grounded-in-human-br.md @@ -0,0 +1,64 @@ +# Paper: RunAgent SuperBrowser: A Theory of Autonomous Web Navigation Grounded in Human Browsing Behaviour + +--- +type: paper +title: "RunAgent SuperBrowser: A Theory of Autonomous Web Navigation Grounded in Human Browsing Behaviour" +authors: Radeen Mostafa, Sawradip Saha +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09399 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, embodied-agent, memory, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09399 diff --git a/papers/items/2026-2606-09426-weavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybr.md b/papers/items/2026-2606-09426-weavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybr.md new file mode 100644 index 0000000..fcc8566 --- /dev/null +++ b/papers/items/2026-2606-09426-weavebench-a-long-horizon-real-world-benchmark-for-computer-use-agents-with-hybr.md @@ -0,0 +1,61 @@ +# Paper: WeaveBench: A Long-Horizon, Real-World Benchmark for Computer-Use Agents with Hybrid Interfaces + +--- +type: paper +title: "WeaveBench: A Long-Horizon, Real-World Benchmark for Computer-Use Agents with Hybrid Interfaces" +authors: Wanli Li, Bowen Zhou, Yunyao Yu, Zhou Xu, Yifan Yang, Dongsheng Li, Caihua Shan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09426 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09426 diff --git a/papers/items/2026-2606-09447-aliyunconsoleagent-training-web-agents-in-real-world-cloud-environments-via-dist.md b/papers/items/2026-2606-09447-aliyunconsoleagent-training-web-agents-in-real-world-cloud-environments-via-dist.md new file mode 100644 index 0000000..d497311 --- /dev/null +++ b/papers/items/2026-2606-09447-aliyunconsoleagent-training-web-agents-in-real-world-cloud-environments-via-dist.md @@ -0,0 +1,61 @@ +# Paper: AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning + +--- +type: paper +title: "AliyunConsoleAgent: Training Web Agents in Real-World Cloud Environments via Distillation and Reinforcement Learning" +authors: Bojie Rong, Zheyu Shen, Qiaoping Wang, Pengfei Kang, Yang Xu, Yawen Wei, Hanyu Wu, Zhi Zhao, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09447 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, rag, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09447 diff --git a/papers/items/2026-2606-09483-memory-beyond-recall-a-dual-process-cognitive-memory-system-for-self-evolving-ll.md b/papers/items/2026-2606-09483-memory-beyond-recall-a-dual-process-cognitive-memory-system-for-self-evolving-ll.md new file mode 100644 index 0000000..8867652 --- /dev/null +++ b/papers/items/2026-2606-09483-memory-beyond-recall-a-dual-process-cognitive-memory-system-for-self-evolving-ll.md @@ -0,0 +1,63 @@ +# Paper: Memory Beyond Recall: A Dual-Process Cognitive Memory System for Self-Evolving LLM Agents + +--- +type: paper +title: "Memory Beyond Recall: A Dual-Process Cognitive Memory System for Self-Evolving LLM Agents" +authors: Tianxiang Fei, Mingyang Song, Mao Zheng, Xiang Yu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09483 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09483 diff --git a/papers/items/2026-2606-09549-secureclaw-clawing-back-control-of-llm-agents.md b/papers/items/2026-2606-09549-secureclaw-clawing-back-control-of-llm-agents.md new file mode 100644 index 0000000..3a555ba --- /dev/null +++ b/papers/items/2026-2606-09549-secureclaw-clawing-back-control-of-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: SecureClaw: Clawing Back Control of LLM Agents + +--- +type: paper +title: "SecureClaw: Clawing Back Control of LLM Agents" +authors: Yuhan Ma, Stefan Schmid +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09549 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09549 diff --git a/papers/items/2026-2606-09738-hdsl-a-hierarchical-domain-specific-language-for-structured-3d-indoor-scene-gene.md b/papers/items/2026-2606-09738-hdsl-a-hierarchical-domain-specific-language-for-structured-3d-indoor-scene-gene.md new file mode 100644 index 0000000..50b40cf --- /dev/null +++ b/papers/items/2026-2606-09738-hdsl-a-hierarchical-domain-specific-language-for-structured-3d-indoor-scene-gene.md @@ -0,0 +1,62 @@ +# Paper: HDSL: A Hierarchical Domain-Specific Language for Structured 3D Indoor Scene Generation and Localized Editing with LLM Agents + +--- +type: paper +title: "HDSL: A Hierarchical Domain-Specific Language for Structured 3D Indoor Scene Generation and Localized Editing with LLM Agents" +authors: Letian Li, Chao Shen, Shuzhao Xie, Chenghao Gu, ZhengXiao He, Yu Meng, Xin Yang, Wenyuan Jiang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09738 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, planning, rag +- arXiv categories: cs.CV +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09738 diff --git a/papers/items/2026-2606-09764-iosworld-a-benchmark-for-personally-intelligent-phone-agents.md b/papers/items/2026-2606-09764-iosworld-a-benchmark-for-personally-intelligent-phone-agents.md new file mode 100644 index 0000000..c6ccf19 --- /dev/null +++ b/papers/items/2026-2606-09764-iosworld-a-benchmark-for-personally-intelligent-phone-agents.md @@ -0,0 +1,62 @@ +# Paper: iOSWorld: A Benchmark for Personally Intelligent Phone Agents + +--- +type: paper +title: "iOSWorld: A Benchmark for Personally Intelligent Phone Agents" +authors: Lawrence Keunho Jang, Mareks Woodside, Geronimo Carom, Andrew Keunwoo Jang, Jing Yu Koh, Ruslan Salakhutdinov +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09764 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, web-gui-agent +- inferred topics: agent-evaluation, computer-use, memory, tool-use +- arXiv categories: cs.LG, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09764 diff --git a/papers/items/2026-2606-09774-auto-configuring-scientific-simulators-with-lightweight-coding-agent-adapters.md b/papers/items/2026-2606-09774-auto-configuring-scientific-simulators-with-lightweight-coding-agent-adapters.md new file mode 100644 index 0000000..0e24092 --- /dev/null +++ b/papers/items/2026-2606-09774-auto-configuring-scientific-simulators-with-lightweight-coding-agent-adapters.md @@ -0,0 +1,65 @@ +# Paper: Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters + +--- +type: paper +title: Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters +authors: Matthew Ho, Brian Liu, Jixuan Chen, Audrey Wang, Lianhui Qin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09774 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, rag, tool-use, world-model +- arXiv categories: cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09774 diff --git a/papers/items/2026-2606-09863-from-confident-closing-to-silent-failure-characterizing-false-success-in-llm-age.md b/papers/items/2026-2606-09863-from-confident-closing-to-silent-failure-characterizing-false-success-in-llm-age.md new file mode 100644 index 0000000..eb906ba --- /dev/null +++ b/papers/items/2026-2606-09863-from-confident-closing-to-silent-failure-characterizing-false-success-in-llm-age.md @@ -0,0 +1,60 @@ +# Paper: From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents + +--- +type: paper +title: "From Confident Closing to Silent Failure: Characterizing False Success in LLM Agents" +authors: Laksh Advani +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09863 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-01 +updated_at: 2026-06-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, tool-use +- arXiv categories: cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09863 diff --git a/papers/items/2026-2606-09961-3spo-state-score-supervised-policy-optimization-for-llm-agents.md b/papers/items/2026-2606-09961-3spo-state-score-supervised-policy-optimization-for-llm-agents.md new file mode 100644 index 0000000..b6cb798 --- /dev/null +++ b/papers/items/2026-2606-09961-3spo-state-score-supervised-policy-optimization-for-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: 3SPO: State-Score-Supervised Policy Optimization for LLM Agents + +--- +type: paper +title: "3SPO: State-Score-Supervised Policy Optimization for LLM Agents" +authors: Yu Han, Kailing Li, Yang Jiao, Yulin Dai, Yuqian Fu, Linhai Zhuo, Tianwen Qian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.09961 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: computer-use, planning, tool-use +- arXiv categories: cs.LG, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.09961 diff --git a/papers/items/2026-2606-10209-less-context-better-agents-efficient-context-engineering-for-long-horizon-tool-u.md b/papers/items/2026-2606-10209-less-context-better-agents-efficient-context-engineering-for-long-horizon-tool-u.md new file mode 100644 index 0000000..4a12c8b --- /dev/null +++ b/papers/items/2026-2606-10209-less-context-better-agents-efficient-context-engineering-for-long-horizon-tool-u.md @@ -0,0 +1,64 @@ +# Paper: Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents + +--- +type: paper +title: "Less Context, Better Agents: Efficient Context Engineering for Long-Horizon Tool-Using LLM Agents" +authors: Abhilasha Lodha, Mahsa Pahlavikhah Varnosfaderani, Abir Chakraborty, Abhinav Mithal +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10209 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-08 +updated_at: 2026-06-08 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.LG, cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10209 diff --git a/papers/items/2026-2606-10304-mirage-a-polarity-flipping-encoding-subspace-in-llm-agents.md b/papers/items/2026-2606-10304-mirage-a-polarity-flipping-encoding-subspace-in-llm-agents.md new file mode 100644 index 0000000..34c9782 --- /dev/null +++ b/papers/items/2026-2606-10304-mirage-a-polarity-flipping-encoding-subspace-in-llm-agents.md @@ -0,0 +1,63 @@ +# Paper: MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents + +--- +type: paper +title: "MIRAGE: A Polarity-Flipping Encoding Subspace in LLM Agents" +authors: Pratibha Revankar, Kargi Chauhan, Jihye Kim, Sadiba Nusrat Nur, Vincent Siu, Chenguang Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10304 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, planning, rag, tool-use +- arXiv categories: cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10304 diff --git a/papers/items/2026-2606-10316-tabclaw-an-interactive-and-self-evolving-agent-for-spreadsheet-manipulation-and-.md b/papers/items/2026-2606-10316-tabclaw-an-interactive-and-self-evolving-agent-for-spreadsheet-manipulation-and-.md new file mode 100644 index 0000000..eb5744f --- /dev/null +++ b/papers/items/2026-2606-10316-tabclaw-an-interactive-and-self-evolving-agent-for-spreadsheet-manipulation-and-.md @@ -0,0 +1,63 @@ +# Paper: TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning + +--- +type: paper +title: "TabClaw: An Interactive and Self-Evolving Agent for Spreadsheet Manipulation and Table Reasoning" +authors: Mingyue Cheng, Shuo Yu, Daoyu Wang, Qingchuan Li, Xiaoyu Tao, Qingyang Mao, Yitong Zhou, Qi Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10316 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, memory, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10316 diff --git a/papers/items/2026-2606-10381-agentic-hybrid-rag-for-evidence-grounded-muon-collider-analysis.md b/papers/items/2026-2606-10381-agentic-hybrid-rag-for-evidence-grounded-muon-collider-analysis.md new file mode 100644 index 0000000..9d70272 --- /dev/null +++ b/papers/items/2026-2606-10381-agentic-hybrid-rag-for-evidence-grounded-muon-collider-analysis.md @@ -0,0 +1,66 @@ +# Paper: Agentic Hybrid RAG for Evidence-Grounded Muon Collider Analysis + +--- +type: paper +title: Agentic Hybrid RAG for Evidence-Grounded Muon Collider Analysis +authors: Ruobing Jiang, Dawei Fu, Cheng Jiang, Tianyi Yang, Zijian Wang, Youpeng Wu, Yong Ban, Yajun Mao, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10381 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - hep-ex + - cs.AI + - cs.CL + - cs.IR + - physics.ins-det +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, rag, reasoning, tool-use, workflow-agent +- arXiv categories: hep-ex, cs.AI, cs.CL, cs.IR, physics.ins-det +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10381 diff --git a/papers/items/2026-2606-10394-stage-claw-automated-state-based-agent-benchmarking-for-realistic-scenarios.md b/papers/items/2026-2606-10394-stage-claw-automated-state-based-agent-benchmarking-for-realistic-scenarios.md new file mode 100644 index 0000000..d1fd181 --- /dev/null +++ b/papers/items/2026-2606-10394-stage-claw-automated-state-based-agent-benchmarking-for-realistic-scenarios.md @@ -0,0 +1,59 @@ +# Paper: STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios + +--- +type: paper +title: "STAGE-Claw: Automated State-based Agent Benchmarking for Realistic Scenarios" +authors: Sirui Liang, Bohan Yu, Peiyu Wang, Shiguang Guo, Wenxing Hu, Pengfei Cao, Jian Zhao, Cao Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10394 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10394 diff --git a/papers/items/2026-2606-10423-webchallenger-a-reliable-and-efficient-generalist-web-agent.md b/papers/items/2026-2606-10423-webchallenger-a-reliable-and-efficient-generalist-web-agent.md new file mode 100644 index 0000000..f6512fb --- /dev/null +++ b/papers/items/2026-2606-10423-webchallenger-a-reliable-and-efficient-generalist-web-agent.md @@ -0,0 +1,63 @@ +# Paper: WebChallenger: A Reliable and Efficient Generalist Web Agent + +--- +type: paper +title: "WebChallenger: A Reliable and Efficient Generalist Web Agent" +authors: Jayoo Hwang, Xiaowen Zhang, Vedant Padwal +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10423 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - computer-use + - embodied-agent + - memory + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: computer-use, embodied-agent, memory, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10423 diff --git a/papers/items/2026-2606-10507-hipif-hierarchical-planning-and-information-folding-for-long-horizon-llm-agent-l.md b/papers/items/2026-2606-10507-hipif-hierarchical-planning-and-information-folding-for-long-horizon-llm-agent-l.md new file mode 100644 index 0000000..ed3f590 --- /dev/null +++ b/papers/items/2026-2606-10507-hipif-hierarchical-planning-and-information-folding-for-long-horizon-llm-agent-l.md @@ -0,0 +1,62 @@ +# Paper: HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning + +--- +type: paper +title: "HIPIF: Hierarchical Planning and Information Folding for Long-Horizon LLM Agent Learning" +authors: Juncheng Diao, Zhicong Lu, Peiguang Li, Yongwei Zhou, Changyuan Tian, Qingbin Li, Rongxiang Weng, Jingang Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10507 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation, autonomous-agent-llm, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, autonomous-agent-llm, planning-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, reasoning +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10507 diff --git a/papers/items/2026-2606-10532-activemem-distributed-active-memory-for-long-horizon-llm-reasoning.md b/papers/items/2026-2606-10532-activemem-distributed-active-memory-for-long-horizon-llm-reasoning.md new file mode 100644 index 0000000..31803ee --- /dev/null +++ b/papers/items/2026-2606-10532-activemem-distributed-active-memory-for-long-horizon-llm-reasoning.md @@ -0,0 +1,62 @@ +# Paper: ActiveMem: Distributed Active Memory for Long-Horizon LLM Reasoning + +--- +type: paper +title: "ActiveMem: Distributed Active Memory for Long-Horizon LLM Reasoning" +authors: Yunhan Jiang, Wenbin Duan, Shasha Guo, Liang Pang, Xiaoqian Sun, Huawei Shen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10532 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-safety + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-safety, memory, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10532 diff --git a/papers/items/2026-2606-10577-agenticnav-zero-shot-vision-and-language-navigation-as-a-tool-calling-harness.md b/papers/items/2026-2606-10577-agenticnav-zero-shot-vision-and-language-navigation-as-a-tool-calling-harness.md new file mode 100644 index 0000000..4b4bf1a --- /dev/null +++ b/papers/items/2026-2606-10577-agenticnav-zero-shot-vision-and-language-navigation-as-a-tool-calling-harness.md @@ -0,0 +1,62 @@ +# Paper: AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness + +--- +type: paper +title: "AgenticNav: Zero-Shot Vision-and-Language Navigation as a Tool-Calling Harness" +authors: Yijian Li, Changze Li, Hantian Shi, Jiaying Luo, Jiyuan Cai, Ming Yang, Tong Qin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10577 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, embodied-agent, memory, rag, tool-use +- arXiv categories: cs.RO +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10577 diff --git a/papers/items/2026-2606-10616-learning-what-to-remember-observability-safe-memory-retention-via-constrained-op.md b/papers/items/2026-2606-10616-learning-what-to-remember-observability-safe-memory-retention-via-constrained-op.md new file mode 100644 index 0000000..af8d5a7 --- /dev/null +++ b/papers/items/2026-2606-10616-learning-what-to-remember-observability-safe-memory-retention-via-constrained-op.md @@ -0,0 +1,63 @@ +# Paper: Learning What to Remember: Observability-Safe Memory Retention via Constrained Optimization for Long-Horizon Language Agents + +--- +type: paper +title: "Learning What to Remember: Observability-Safe Memory Retention via Constrained Optimization for Long-Horizon Language Agents" +authors: Qingcan Kang, Liu Mingyang, Shixiong Kai, Kaichao Liang, Tao Zhong, Mingxuan Yuan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10616 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10616 diff --git a/papers/items/2026-2606-10677-infini-memory-maintainable-topic-documents-for-long-term-llm-agent-memory.md b/papers/items/2026-2606-10677-infini-memory-maintainable-topic-documents-for-long-term-llm-agent-memory.md new file mode 100644 index 0000000..062e995 --- /dev/null +++ b/papers/items/2026-2606-10677-infini-memory-maintainable-topic-documents-for-long-term-llm-agent-memory.md @@ -0,0 +1,62 @@ +# Paper: Infini Memory: Maintainable Topic Documents for Long-Term LLM Agent Memory + +--- +type: paper +title: "Infini Memory: Maintainable Topic Documents for Long-Term LLM Agent Memory" +authors: Suozhao Ji, Baodong Wu, Zehao Wang, Lei Xia, Qingping Li, Ruisong Wang, Wenbo Ding, Zhenhua Zhu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10677 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10677 diff --git a/papers/items/2026-2606-10684-divide-and-cooperate-role-decomposed-multi-agent-llm-training-with-cross-agent-l.md b/papers/items/2026-2606-10684-divide-and-cooperate-role-decomposed-multi-agent-llm-training-with-cross-agent-l.md new file mode 100644 index 0000000..f14f7b8 --- /dev/null +++ b/papers/items/2026-2606-10684-divide-and-cooperate-role-decomposed-multi-agent-llm-training-with-cross-agent-l.md @@ -0,0 +1,62 @@ +# Paper: Divide and Cooperate: Role-Decomposed Multi-Agent LLM Training with Cross-Agent Learning Signals + +--- +type: paper +title: "Divide and Cooperate: Role-Decomposed Multi-Agent LLM Training with Cross-Agent Learning Signals" +authors: Jaewan Park, Solbee Cho, Jay-Yoon Lee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10684 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, multi-agent, reasoning, tool-use +- arXiv categories: cs.LG, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10684 diff --git a/papers/items/2026-2606-10742-memvenom-triggered-poisoning-of-multimodal-memories-in-web-agents.md b/papers/items/2026-2606-10742-memvenom-triggered-poisoning-of-multimodal-memories-in-web-agents.md new file mode 100644 index 0000000..3bf1c73 --- /dev/null +++ b/papers/items/2026-2606-10742-memvenom-triggered-poisoning-of-multimodal-memories-in-web-agents.md @@ -0,0 +1,64 @@ +# Paper: MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents + +--- +type: paper +title: "MemVenom: Triggered Poisoning of Multimodal Memories in Web Agents" +authors: Yv Zhang, Hao Sun, Hao Fang, Kuofeng Gao, Fan Mo, Bin Chen, Shu-Tao Xia, Yaowei Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10742 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, rag, reasoning +- arXiv categories: cs.CR, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10742 diff --git a/papers/items/2026-2606-10749-toward-secure-llm-agents-threat-surfaces-attacks-defenses-and-evaluation.md b/papers/items/2026-2606-10749-toward-secure-llm-agents-threat-surfaces-attacks-defenses-and-evaluation.md new file mode 100644 index 0000000..ac102e1 --- /dev/null +++ b/papers/items/2026-2606-10749-toward-secure-llm-agents-threat-surfaces-attacks-defenses-and-evaluation.md @@ -0,0 +1,65 @@ +# Paper: Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation + +--- +type: paper +title: "Toward Secure LLM Agents: Threat Surfaces, Attacks, Defenses, and Evaluation" +authors: Yuchen Ling, Shengcheng Yu, Zhenyu Chen, Chunrong Fang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10749 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - multi-agent + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 25 +collection_queries: agent-safety, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, planning-agent +- inferred topics: agent-evaluation, agent-safety, memory, multi-agent, planning, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 25 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10749 diff --git a/papers/items/2026-2606-10921-trace-only-what-you-need-structure-aware-on-demand-hypergraph-memory-for-long-do.md b/papers/items/2026-2606-10921-trace-only-what-you-need-structure-aware-on-demand-hypergraph-memory-for-long-do.md new file mode 100644 index 0000000..c3f196a --- /dev/null +++ b/papers/items/2026-2606-10921-trace-only-what-you-need-structure-aware-on-demand-hypergraph-memory-for-long-do.md @@ -0,0 +1,64 @@ +# Paper: Trace Only What You Need: Structure-Aware On-Demand Hypergraph Memory for Long-Document Question Answering + +--- +type: paper +title: "Trace Only What You Need: Structure-Aware On-Demand Hypergraph Memory for Long-Document Question Answering" +authors: Xiangjun Zai, Xingyu Tan, Chen Chen, Xiaoyang Wang, Wenjie Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10921 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - multi-agent + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, computer-use, memory, multi-agent, planning, rag, reasoning +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10921 diff --git a/papers/items/2026-2606-10933-frontier-coding-agents-use-metaprogramming-to-adapt-to-unfamiliar-programming-la.md b/papers/items/2026-2606-10933-frontier-coding-agents-use-metaprogramming-to-adapt-to-unfamiliar-programming-la.md new file mode 100644 index 0000000..5e596b6 --- /dev/null +++ b/papers/items/2026-2606-10933-frontier-coding-agents-use-metaprogramming-to-adapt-to-unfamiliar-programming-la.md @@ -0,0 +1,61 @@ +# Paper: Frontier Coding Agents Use Metaprogramming to Adapt to Unfamiliar Programming Languages + +--- +type: paper +title: Frontier Coding Agents Use Metaprogramming to Adapt to Unfamiliar Programming Languages +authors: Aman Sharma, Sushrut Thorat, Paras Chopra +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.10933 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, computer-use, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.10933 diff --git a/papers/items/2026-2606-11042-workflow-gym-towards-long-horizon-evaluation-of-computer-use-agentic-tasks-in-re.md b/papers/items/2026-2606-11042-workflow-gym-towards-long-horizon-evaluation-of-computer-use-agentic-tasks-in-re.md new file mode 100644 index 0000000..bedaa64 --- /dev/null +++ b/papers/items/2026-2606-11042-workflow-gym-towards-long-horizon-evaluation-of-computer-use-agentic-tasks-in-re.md @@ -0,0 +1,62 @@ +# Paper: Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields + +--- +type: paper +title: "Workflow-GYM: Towards Long-Horizon Evaluation of Computer-use Agentic tasks in Real-World Professional Fields" +authors: Liya Zhu, Jingzhe Ding, Jian Zhang, Jianbo Xue, Shihao Liang, Ge Zhang, Yi Zhu, Duju Zeng, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11042 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11042 diff --git a/papers/items/2026-2606-11078-a-history-aware-visually-grounded-critic-for-computer-use-agents.md b/papers/items/2026-2606-11078-a-history-aware-visually-grounded-critic-for-computer-use-agents.md new file mode 100644 index 0000000..04af856 --- /dev/null +++ b/papers/items/2026-2606-11078-a-history-aware-visually-grounded-critic-for-computer-use-agents.md @@ -0,0 +1,64 @@ +# Paper: A History-Aware Visually Grounded Critic for Computer Use Agents + +--- +type: paper +title: A History-Aware Visually Grounded Critic for Computer Use Agents +authors: Jaewoo Lee, Zaid Khan, Archiki Prasad, Justin Chih-Yao Chen, Supriyo Chakraborty, Kartik Balasubramaniam, Sambit Sahu, Elias Stengel-Eskin, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11078 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, rag, tool-use +- arXiv categories: cs.AI, cs.CL, cs.CV +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11078 diff --git a/papers/items/2026-2606-11079-vista-a-versatile-interactive-user-simulation-toolkit-for-agent-evaluation.md b/papers/items/2026-2606-11079-vista-a-versatile-interactive-user-simulation-toolkit-for-agent-evaluation.md new file mode 100644 index 0000000..b1a3c83 --- /dev/null +++ b/papers/items/2026-2606-11079-vista-a-versatile-interactive-user-simulation-toolkit-for-agent-evaluation.md @@ -0,0 +1,61 @@ +# Paper: VISTA: A Versatile Interactive User Simulation Toolkit for Agent Evaluation + +--- +type: paper +title: "VISTA: A Versatile Interactive User Simulation Toolkit for Agent Evaluation" +authors: Yunan Lu, Ryan Shea, Yusen Zhang, Zhou Yu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11079 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, rag, tool-use, world-model +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11079 diff --git a/papers/items/2026-2606-11119-trace-a-unified-rollout-budget-allocation-framework-for-efficient-agentic-reinfo.md b/papers/items/2026-2606-11119-trace-a-unified-rollout-budget-allocation-framework-for-efficient-agentic-reinfo.md new file mode 100644 index 0000000..dc1fde5 --- /dev/null +++ b/papers/items/2026-2606-11119-trace-a-unified-rollout-budget-allocation-framework-for-efficient-agentic-reinfo.md @@ -0,0 +1,64 @@ +# Paper: TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning + +--- +type: paper +title: "TRACE: A Unified Rollout Budget Allocation Framework for Efficient Agentic Reinforcement Learning" +authors: Heming Zou, Qi Wang, Yun Qu, Yuhang Jiang, Lizhou Cai, Yixiu Mao, Ru Peng, Xin Xu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11119 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, computer-use, rag, reasoning, tool-use +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11119 diff --git a/papers/items/2026-2606-11176-data-journalist-agent-transforming-data-into-verifiable-multimodal-stories.md b/papers/items/2026-2606-11176-data-journalist-agent-transforming-data-into-verifiable-multimodal-stories.md new file mode 100644 index 0000000..5df92da --- /dev/null +++ b/papers/items/2026-2606-11176-data-journalist-agent-transforming-data-into-verifiable-multimodal-stories.md @@ -0,0 +1,66 @@ +# Paper: Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories + +--- +type: paper +title: "Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories" +authors: Kevin Qinghong Lin, Batu EI, Yuhong Shi, Pan Lu, Philip Torr, James Zou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11176 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.CL + - cs.CY + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, coding-agent, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.CV, cs.CL, cs.CY, cs.HC +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11176 diff --git a/papers/items/2026-2606-11349-knowing-when-to-ask-self-gated-clarification-for-hierarchical-language-agents.md b/papers/items/2026-2606-11349-knowing-when-to-ask-self-gated-clarification-for-hierarchical-language-agents.md new file mode 100644 index 0000000..a57230f --- /dev/null +++ b/papers/items/2026-2606-11349-knowing-when-to-ask-self-gated-clarification-for-hierarchical-language-agents.md @@ -0,0 +1,62 @@ +# Paper: Knowing When to Ask: Self-Gated Clarification for Hierarchical Language Agents + +--- +type: paper +title: "Knowing When to Ask: Self-Gated Clarification for Hierarchical Language Agents" +authors: Aijing Gao, Yiming Kang, Mengdie Flora Wang, Jae Oh Woo +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11349 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, embodied-agent, reasoning, tool-use +- arXiv categories: cs.AI, cs.HC +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11349 diff --git a/papers/items/2026-2606-11354-a-zero-shot-multi-agent-framework-for-human-building-interaction-via-programmati.md b/papers/items/2026-2606-11354-a-zero-shot-multi-agent-framework-for-human-building-interaction-via-programmati.md new file mode 100644 index 0000000..b68a7a9 --- /dev/null +++ b/papers/items/2026-2606-11354-a-zero-shot-multi-agent-framework-for-human-building-interaction-via-programmati.md @@ -0,0 +1,64 @@ +# Paper: A Zero-Shot Multi-Agent Framework for Human-Building Interaction via Programmatic Reasoning + +--- +type: paper +title: A Zero-Shot Multi-Agent Framework for Human-Building Interaction via Programmatic Reasoning +authors: Yuqi Wang, Gulai Shen, Ali Mehmani +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11354 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-09 +updated_at: 2026-06-09 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - multi-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.ET +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-safety, coding-agent, multi-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.ET +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11354 diff --git a/papers/items/2026-2606-11680-organize-then-retrieve-hierarchical-memory-navigation-for-efficient-agents.md b/papers/items/2026-2606-11680-organize-then-retrieve-hierarchical-memory-navigation-for-efficient-agents.md new file mode 100644 index 0000000..bd295c1 --- /dev/null +++ b/papers/items/2026-2606-11680-organize-then-retrieve-hierarchical-memory-navigation-for-efficient-agents.md @@ -0,0 +1,66 @@ +# Paper: Organize then Retrieve: Hierarchical Memory Navigation for Efficient Agents + +--- +type: paper +title: "Organize then Retrieve: Hierarchical Memory Navigation for Efficient Agents" +authors: Hao-Lun Hsu, Nikki Lijing Kuang, Boyi Liu, Zhewei Yao, Yuxiong He +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11680 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - memory + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, embodied-agent, memory, planning, rag, reasoning +- arXiv categories: cs.AI, cs.CL, cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11680 diff --git a/papers/items/2026-2606-11688-goal-autopilot-a-verifiable-anti-fabrication-firewall-for-unattended-long-horizo.md b/papers/items/2026-2606-11688-goal-autopilot-a-verifiable-anti-fabrication-firewall-for-unattended-long-horizo.md new file mode 100644 index 0000000..5f55e22 --- /dev/null +++ b/papers/items/2026-2606-11688-goal-autopilot-a-verifiable-anti-fabrication-firewall-for-unattended-long-horizo.md @@ -0,0 +1,62 @@ +# Paper: Goal-Autopilot: A Verifiable Anti-Fabrication Firewall for Unattended Long-Horizon Agents + +--- +type: paper +title: "Goal-Autopilot: A Verifiable Anti-Fabrication Firewall for Unattended Long-Horizon Agents" +authors: Youwang Deng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11688 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, coding-agent, planning, rag +- arXiv categories: cs.CL, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11688 diff --git a/papers/items/2026-2606-11702-medcta-a-benchmark-for-clinical-tool-agents.md b/papers/items/2026-2606-11702-medcta-a-benchmark-for-clinical-tool-agents.md new file mode 100644 index 0000000..e3972c5 --- /dev/null +++ b/papers/items/2026-2606-11702-medcta-a-benchmark-for-clinical-tool-agents.md @@ -0,0 +1,63 @@ +# Paper: MedCTA: A Benchmark for Clinical Tool Agents + +--- +type: paper +title: "MedCTA: A Benchmark for Clinical Tool Agents" +authors: Tajamul Ashraf, Hyewon Jeong, Fida Mohammad Thoker, Bernard Ghanem +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11702 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, planning, rag, tool-use +- arXiv categories: cs.CV, cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11702 diff --git a/papers/items/2026-2606-11869-agents-all-the-way-down-a-methodology-for-building-custom-ai-agents-from-substra.md b/papers/items/2026-2606-11869-agents-all-the-way-down-a-methodology-for-building-custom-ai-agents-from-substra.md new file mode 100644 index 0000000..672f172 --- /dev/null +++ b/papers/items/2026-2606-11869-agents-all-the-way-down-a-methodology-for-building-custom-ai-agents-from-substra.md @@ -0,0 +1,62 @@ +# Paper: Agents All the Way Down; A Methodology for Building Custom AI Agents from Substrate to Production + +--- +type: paper +title: Agents All the Way Down; A Methodology for Building Custom AI Agents from Substrate to Production +authors: Marc Alier Forment, Juanan Pereira, Francisco José García-Peñalvo, María José Casañ Guerrero +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.11869 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-safety, coding-agent, multi-agent, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.11869 diff --git a/papers/items/2026-2606-12195-internvideo3-agentify-foundation-models-with-multimodal-contextual-reasoning.md b/papers/items/2026-2606-12195-internvideo3-agentify-foundation-models-with-multimodal-contextual-reasoning.md new file mode 100644 index 0000000..96f8adc --- /dev/null +++ b/papers/items/2026-2606-12195-internvideo3-agentify-foundation-models-with-multimodal-contextual-reasoning.md @@ -0,0 +1,63 @@ +# Paper: InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning + +--- +type: paper +title: "InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning" +authors: Ziang Yan, Sheng Xia, Jiashuo Yu, Yue Wu, Tianxiang Jiang, Songze Li, Kanghui Tian, Yicheng Xu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12195 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.CV +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12195 diff --git a/papers/items/2026-2606-12320-a-five-plane-reference-architecture-for-runtime-governance-of-production-ai-agen.md b/papers/items/2026-2606-12320-a-five-plane-reference-architecture-for-runtime-governance-of-production-ai-agen.md new file mode 100644 index 0000000..24519e3 --- /dev/null +++ b/papers/items/2026-2606-12320-a-five-plane-reference-architecture-for-runtime-governance-of-production-ai-agen.md @@ -0,0 +1,67 @@ +# Paper: A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents + +--- +type: paper +title: A Five-Plane Reference Architecture for Runtime Governance of Production AI Agents +authors: Krti Tallam +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12320 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CC + - cs.CR + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, memory, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CC, cs.CR, cs.SE +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12320 diff --git a/papers/items/2026-2606-12341-ocelot-inference-leakage-budgets-for-privacy-preserving-llm-agents.md b/papers/items/2026-2606-12341-ocelot-inference-leakage-budgets-for-privacy-preserving-llm-agents.md new file mode 100644 index 0000000..57d3ea8 --- /dev/null +++ b/papers/items/2026-2606-12341-ocelot-inference-leakage-budgets-for-privacy-preserving-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents + +--- +type: paper +title: "OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents" +authors: Jin Xie, Songze Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12341 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, reasoning, tool-use +- arXiv categories: cs.CR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12341 diff --git a/papers/items/2026-2606-12344-claw-swe-bench-a-benchmark-for-evaluating-openclaw-style-agent-harnesses-on-codi.md b/papers/items/2026-2606-12344-claw-swe-bench-a-benchmark-for-evaluating-openclaw-style-agent-harnesses-on-codi.md new file mode 100644 index 0000000..5d10d73 --- /dev/null +++ b/papers/items/2026-2606-12344-claw-swe-bench-a-benchmark-for-evaluating-openclaw-style-agent-harnesses-on-codi.md @@ -0,0 +1,61 @@ +# Paper: Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks + +--- +type: paper +title: "Claw-SWE-Bench: A Benchmark for Evaluating OpenClaw-style Agent Harnesses on Coding Tasks" +authors: Mengyu Zheng, Kai Han, Boxun Li, Haiyang Xu, Yuchuan Tian, Wei He, Hang Zhou, Jianyuan Guo, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12344 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, tool-use +- arXiv categories: cs.LG, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12344 diff --git a/papers/items/2026-2606-12384-appo-agentic-procedural-policy-optimization.md b/papers/items/2026-2606-12384-appo-agentic-procedural-policy-optimization.md new file mode 100644 index 0000000..d0227ed --- /dev/null +++ b/papers/items/2026-2606-12384-appo-agentic-procedural-policy-optimization.md @@ -0,0 +1,61 @@ +# Paper: APPO: Agentic Procedural Policy Optimization + +--- +type: paper +title: "APPO: Agentic Procedural Policy Optimization" +authors: Xucong Wang, Ziyu Ma, Yong Wang, Yuxiang Ji, Shidong Yang, Guanhua Chen, Pengkun Wang, Xiangxiang Chu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12384 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, tool-use, workflow-agent +- arXiv categories: cs.LG, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12384 diff --git a/papers/items/2026-2606-12563-arbor-tree-search-as-a-cognition-layer-for-autonomous-agents.md b/papers/items/2026-2606-12563-arbor-tree-search-as-a-cognition-layer-for-autonomous-agents.md new file mode 100644 index 0000000..ef8d0e9 --- /dev/null +++ b/papers/items/2026-2606-12563-arbor-tree-search-as-a-cognition-layer-for-autonomous-agents.md @@ -0,0 +1,61 @@ +# Paper: Arbor: Tree Search as a Cognition Layer for Autonomous Agents + +--- +type: paper +title: "Arbor: Tree Search as a Cognition Layer for Autonomous Agents" +authors: Neha Prakriya, Chaojun Hou, Zheng Gong, Huasha Zhao, Xi Zhao, Mou Li, Zhenyu Gu, Emad Barsoum +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12563 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, memory, multi-agent, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12563 diff --git a/papers/items/2026-2606-12586-beyond-attack-success-rate-examining-trigger-leakage-in-vision-language-agentic-.md b/papers/items/2026-2606-12586-beyond-attack-success-rate-examining-trigger-leakage-in-vision-language-agentic-.md new file mode 100644 index 0000000..96b1d51 --- /dev/null +++ b/papers/items/2026-2606-12586-beyond-attack-success-rate-examining-trigger-leakage-in-vision-language-agentic-.md @@ -0,0 +1,62 @@ +# Paper: Beyond Attack Success Rate: Examining Trigger Leakage in Vision-Language Agentic Systems + +--- +type: paper +title: "Beyond Attack Success Rate: Examining Trigger Leakage in Vision-Language Agentic Systems" +authors: Jiamin Chang, Salil Kanhere, Piotr Koniusz, Jason, Xue, Hammond Pearce +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12586 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent, tool-use +- inferred topics: agent-evaluation, embodied-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.CR +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12586 diff --git a/papers/items/2026-2606-12634-keep-policy-gradient-in-charge-sibling-guided-credit-distillation-for-long-horiz.md b/papers/items/2026-2606-12634-keep-policy-gradient-in-charge-sibling-guided-credit-distillation-for-long-horiz.md new file mode 100644 index 0000000..842fd4a --- /dev/null +++ b/papers/items/2026-2606-12634-keep-policy-gradient-in-charge-sibling-guided-credit-distillation-for-long-horiz.md @@ -0,0 +1,64 @@ +# Paper: Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents + +--- +type: paper +title: "Keep Policy Gradient in Charge: Sibling-Guided Credit Distillation for Long-Horizon Tool-Use Agents" +authors: Tianyu Ding, Jianhong Xin, Juan Pablo De la Cruz Weinstein +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12634 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, computer-use, planning, reasoning, tool-use +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12634 diff --git a/papers/items/2026-2606-12657-trajgenagent-a-hierarchical-llm-agent-for-human-mobility-trajectory-generation.md b/papers/items/2026-2606-12657-trajgenagent-a-hierarchical-llm-agent-for-human-mobility-trajectory-generation.md new file mode 100644 index 0000000..a7fc4fa --- /dev/null +++ b/papers/items/2026-2606-12657-trajgenagent-a-hierarchical-llm-agent-for-human-mobility-trajectory-generation.md @@ -0,0 +1,65 @@ +# Paper: TrajGenAgent: A Hierarchical LLM Agent for Human Mobility Trajectory Generation + +--- +type: paper +title: "TrajGenAgent: A Hierarchical LLM Agent for Human Mobility Trajectory Generation" +authors: Siyu Li, Toan Tran, Lingyi Zhao, Khurram Shafique, Li Xiong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12657 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.DB + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, workflow-agent, world-model +- arXiv categories: cs.AI, cs.DB, cs.RO +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12657 diff --git a/papers/items/2026-2606-12674-evoflux-inference-time-evolution-of-executable-tool-workflows-for-compact-agents.md b/papers/items/2026-2606-12674-evoflux-inference-time-evolution-of-executable-tool-workflows-for-compact-agents.md new file mode 100644 index 0000000..6cf5ffa --- /dev/null +++ b/papers/items/2026-2606-12674-evoflux-inference-time-evolution-of-executable-tool-workflows-for-compact-agents.md @@ -0,0 +1,62 @@ +# Paper: Evoflux: Inference-Time Evolution of Executable Tool Workflows for Compact Agents + +--- +type: paper +title: "Evoflux: Inference-Time Evolution of Executable Tool Workflows for Compact Agents" +authors: Kushal Raj Bhandari, Ling Yue, Ching-Yun Ko, Dhaval Patel, Shaowu Pan, Pin-Yu Chen, Jianxi Gao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12674 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: function-calling, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling, tool-use +- inferred topics: agent-safety, computer-use, planning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12674 diff --git a/papers/items/2026-2606-12703-smsr-certified-defence-against-runtime-memory-poisoning-in-persistent-llm-agent-.md b/papers/items/2026-2606-12703-smsr-certified-defence-against-runtime-memory-poisoning-in-persistent-llm-agent-.md new file mode 100644 index 0000000..407fe1a --- /dev/null +++ b/papers/items/2026-2606-12703-smsr-certified-defence-against-runtime-memory-poisoning-in-persistent-llm-agent-.md @@ -0,0 +1,63 @@ +# Paper: SMSR: Certified Defence Against Runtime Memory Poisoning in Persistent LLM Agent Systems + +--- +type: paper +title: "SMSR: Certified Defence Against Runtime Memory Poisoning in Persistent LLM Agent Systems" +authors: Tarun Sharma +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12703 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, memory, rag, workflow-agent +- arXiv categories: cs.CR, cs.AI, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12703 diff --git a/papers/items/2026-2606-12780-proplay-procedural-world-models-for-self-evolving-llm-agents.md b/papers/items/2026-2606-12780-proplay-procedural-world-models-for-self-evolving-llm-agents.md new file mode 100644 index 0000000..0c2c539 --- /dev/null +++ b/papers/items/2026-2606-12780-proplay-procedural-world-models-for-self-evolving-llm-agents.md @@ -0,0 +1,64 @@ +# Paper: ProPlay: Procedural World Models for Self-Evolving LLM Agents + +--- +type: paper +title: "ProPlay: Procedural World Models for Self-Evolving LLM Agents" +authors: Yijun Ma, Zehong Wang, Yiyang Li, Ziming Li, Xiaoguang Guo, Weixiang Sun, Chuxu Zhang, Yanfang Ye +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12780 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, tool-use, world-model +- arXiv categories: cs.LG, cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12780 diff --git a/papers/items/2026-2606-12837-lohosearch-benchmarking-long-horizon-search-agents-beyond-the-human-difficulty-c.md b/papers/items/2026-2606-12837-lohosearch-benchmarking-long-horizon-search-agents-beyond-the-human-difficulty-c.md new file mode 100644 index 0000000..0b82351 --- /dev/null +++ b/papers/items/2026-2606-12837-lohosearch-benchmarking-long-horizon-search-agents-beyond-the-human-difficulty-c.md @@ -0,0 +1,61 @@ +# Paper: LoHoSearch: Benchmarking Long-Horizon Search Agents Beyond the Human Difficulty Ceiling + +--- +type: paper +title: "LoHoSearch: Benchmarking Long-Horizon Search Agents Beyond the Human Difficulty Ceiling" +authors: Jiarui Zhao, Rongzhi Zhang, Lingchuan Liu, Hao Yang, Xunliang Cai, Xi Su +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12837 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, planning, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12837 diff --git a/papers/items/2026-2606-12945-learning-what-to-remember-a-cognitively-grounded-multi-factor-value-model-for-ag.md b/papers/items/2026-2606-12945-learning-what-to-remember-a-cognitively-grounded-multi-factor-value-model-for-ag.md new file mode 100644 index 0000000..523d13d --- /dev/null +++ b/papers/items/2026-2606-12945-learning-what-to-remember-a-cognitively-grounded-multi-factor-value-model-for-ag.md @@ -0,0 +1,64 @@ +# Paper: Learning What to Remember: A Cognitively Grounded Multi-Factor Value Model for Agentic Memory + +--- +type: paper +title: "Learning What to Remember: A Cognitively Grounded Multi-Factor Value Model for Agentic Memory" +authors: Zhibao Chen, Qian Cheng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.12945 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.12945 diff --git a/papers/items/2026-2606-13148-terrabench-can-agents-reason-over-heterogeneous-earth-system-data.md b/papers/items/2026-2606-13148-terrabench-can-agents-reason-over-heterogeneous-earth-system-data.md new file mode 100644 index 0000000..9195c80 --- /dev/null +++ b/papers/items/2026-2606-13148-terrabench-can-agents-reason-over-heterogeneous-earth-system-data.md @@ -0,0 +1,64 @@ +# Paper: TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data? + +--- +type: paper +title: "TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?" +authors: Dat Tien Nguyen, Thao Nguyen, Fadillah Adamsyah Maani, Huy M. Le, Muhammad Umer Sheikh, Numan Saeed, Muhammad Haris Khan, Salman Khan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13148 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use, workflow-agent, world-model +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13148 diff --git a/papers/items/2026-2606-13177-memrefine-llm-guided-compression-for-long-term-agent-memory.md b/papers/items/2026-2606-13177-memrefine-llm-guided-compression-for-long-term-agent-memory.md new file mode 100644 index 0000000..63ffb88 --- /dev/null +++ b/papers/items/2026-2606-13177-memrefine-llm-guided-compression-for-long-term-agent-memory.md @@ -0,0 +1,64 @@ +# Paper: MemRefine: LLM-Guided Compression for Long-Term Agent Memory + +--- +type: paper +title: "MemRefine: LLM-Guided Compression for Long-Term Agent Memory" +authors: Minjae Kim, Jinheon Baek, Soyeong Jeong, Sung Ju Hwang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13177 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, rag, tool-use +- arXiv categories: cs.CL, cs.AI, cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13177 diff --git a/papers/items/2026-2606-13192-reasoning-for-mobile-user-experience-with-multimodal-llms-task-benchmark-and-app.md b/papers/items/2026-2606-13192-reasoning-for-mobile-user-experience-with-multimodal-llms-task-benchmark-and-app.md new file mode 100644 index 0000000..a422e35 --- /dev/null +++ b/papers/items/2026-2606-13192-reasoning-for-mobile-user-experience-with-multimodal-llms-task-benchmark-and-app.md @@ -0,0 +1,62 @@ +# Paper: Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach + +--- +type: paper +title: "Reasoning for Mobile User Experience with Multimodal LLMs: Task, Benchmark, and Approach" +authors: Ruichao Mao, Zhou Fang, Teng Guo, Hao Yang, Yaping Li, Shaohua Peng, Maji Huang, Xiaoyu Lin, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13192 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13192 diff --git a/papers/items/2026-2606-13317-skillcat-contrastive-assessment-and-topology-aware-skill-self-evolution-for-llm-.md b/papers/items/2026-2606-13317-skillcat-contrastive-assessment-and-topology-aware-skill-self-evolution-for-llm-.md new file mode 100644 index 0000000..7176efb --- /dev/null +++ b/papers/items/2026-2606-13317-skillcat-contrastive-assessment-and-topology-aware-skill-self-evolution-for-llm-.md @@ -0,0 +1,60 @@ +# Paper: SkillCAT: Contrastive Assessment and Topology-Aware Skill Self-Evolution for LLM Agents + +--- +type: paper +title: "SkillCAT: Contrastive Assessment and Topology-Aware Skill Self-Evolution for LLM Agents" +authors: Kunfeng Chen, Qihuang Zhong, Juhua Liu, Bo Du +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13317 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, rag, tool-use +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13317 diff --git a/papers/items/2026-2606-13385-who-pays-the-price-stakeholder-centric-prompt-injection-benchmarking-for-real-wo.md b/papers/items/2026-2606-13385-who-pays-the-price-stakeholder-centric-prompt-injection-benchmarking-for-real-wo.md new file mode 100644 index 0000000..5568ab3 --- /dev/null +++ b/papers/items/2026-2606-13385-who-pays-the-price-stakeholder-centric-prompt-injection-benchmarking-for-real-wo.md @@ -0,0 +1,65 @@ +# Paper: Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents + +--- +type: paper +title: Who Pays the Price? Stakeholder-Centric Prompt Injection Benchmarking for Real-world Web Agents +authors: Zihao Wang, Yiming Li, Yutong Wu, Zheyu Liu, Kangjie Chen, Fok Kar Wai, Pin-Yu Chen, Vrizlynn L. L. Thing, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13385 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.CY + - cs.HC + - cs.MM +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, tool-use +- arXiv categories: cs.CR, cs.AI, cs.CY, cs.HC, cs.MM +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13385 diff --git a/papers/items/2026-2606-13602-epibench-verifiable-evaluation-of-ai-agents-on-epigenomics-analysis.md b/papers/items/2026-2606-13602-epibench-verifiable-evaluation-of-ai-agents-on-epigenomics-analysis.md new file mode 100644 index 0000000..dcfc0f2 --- /dev/null +++ b/papers/items/2026-2606-13602-epibench-verifiable-evaluation-of-ai-agents-on-epigenomics-analysis.md @@ -0,0 +1,59 @@ +# Paper: EpiBench: Verifiable Evaluation of AI Agents on Epigenomics Analysis + +--- +type: paper +title: "EpiBench: Verifiable Evaluation of AI Agents on Epigenomics Analysis" +authors: Harihara Muralidharan, Reema Baskar, Soo Hee Lee, Tim Proctor, Kenny Workman +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13602 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, workflow-agent +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13602 diff --git a/papers/items/2026-2606-13608-agentbeats-agentifying-agent-assessment-for-openness-standardization-and-reprodu.md b/papers/items/2026-2606-13608-agentbeats-agentifying-agent-assessment-for-openness-standardization-and-reprodu.md new file mode 100644 index 0000000..44e1e5f --- /dev/null +++ b/papers/items/2026-2606-13608-agentbeats-agentifying-agent-assessment-for-openness-standardization-and-reprodu.md @@ -0,0 +1,63 @@ +# Paper: AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility + +--- +type: paper +title: "AgentBeats: Agentifying Agent Assessment for Openness, Standardization, and Reproducibility" +authors: Xiaoyuan Liu, Jianhong Tu, Yuqi Chen, Siyuan Xie, Sihan Ren, Tianneng Shi, Gal Gantar, Evan Sandoval, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13608 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-14 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, multi-agent, rag, tool-use +- arXiv categories: cs.AI, cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13608 diff --git a/papers/items/2026-2606-13643-recursive-agent-harnesses.md b/papers/items/2026-2606-13643-recursive-agent-harnesses.md new file mode 100644 index 0000000..82f1a33 --- /dev/null +++ b/papers/items/2026-2606-13643-recursive-agent-harnesses.md @@ -0,0 +1,63 @@ +# Paper: Recursive Agent Harnesses + +--- +type: paper +title: Recursive Agent Harnesses +authors: Elias Lumer, Sahil Sen, Kevin Paul, Vamse Kumar Subbiah +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13643 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling +- inferred topics: agent-evaluation, coding-agent, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13643 diff --git a/papers/items/2026-2606-13663-hypertool-beyond-step-wise-tool-calls-for-tool-augmented-agents.md b/papers/items/2026-2606-13663-hypertool-beyond-step-wise-tool-calls-for-tool-augmented-agents.md new file mode 100644 index 0000000..b151d94 --- /dev/null +++ b/papers/items/2026-2606-13663-hypertool-beyond-step-wise-tool-calls-for-tool-augmented-agents.md @@ -0,0 +1,61 @@ +# Paper: HyperTool: Beyond Step-Wise Tool Calls for Tool-Augmented Agents + +--- +type: paper +title: "HyperTool: Beyond Step-Wise Tool Calls for Tool-Augmented Agents" +authors: Yaxin Du, Yifan Zhou, Yujie Ge, Jiajun Wang, Xianghe Pang, Shuo Tang, Tuney Zheng, Bryan Dai, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13663 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13663 diff --git a/papers/items/2026-2606-13686-benchmarking-web-agent-safety-under-e-commerce-deceptive-interfaces.md b/papers/items/2026-2606-13686-benchmarking-web-agent-safety-under-e-commerce-deceptive-interfaces.md new file mode 100644 index 0000000..df750e4 --- /dev/null +++ b/papers/items/2026-2606-13686-benchmarking-web-agent-safety-under-e-commerce-deceptive-interfaces.md @@ -0,0 +1,61 @@ +# Paper: Benchmarking Web Agent Safety under E-commerce Deceptive Interfaces + +--- +type: paper +title: Benchmarking Web Agent Safety under E-commerce Deceptive Interfaces +authors: Zijing Shi, Meng Fang, Ling Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13686 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-26 +updated_at: 2026-04-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.CY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, computer-use +- arXiv categories: cs.CL, cs.CY +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13686 diff --git a/papers/items/2026-2606-13904-sana-what-matters-for-qa-agents-over-massive-data-lakes.md b/papers/items/2026-2606-13904-sana-what-matters-for-qa-agents-over-massive-data-lakes.md new file mode 100644 index 0000000..94b6935 --- /dev/null +++ b/papers/items/2026-2606-13904-sana-what-matters-for-qa-agents-over-massive-data-lakes.md @@ -0,0 +1,64 @@ +# Paper: SANA: What Matters for QA Agents over Massive Data Lakes? + +--- +type: paper +title: "SANA: What Matters for QA Agents over Massive Data Lakes?" +authors: Austin Senna Wijaya, Jiaxiang Liu, Haonan Wang, Eugene Wu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13904 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.DB +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, embodied-agent, planning, tool-use +- arXiv categories: cs.CL, cs.AI, cs.DB +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13904 diff --git a/papers/items/2026-2606-13994-hidden-in-plain-sight-benchmarking-agent-safety-against-decomposition-attacks-wi.md b/papers/items/2026-2606-13994-hidden-in-plain-sight-benchmarking-agent-safety-against-decomposition-attacks-wi.md new file mode 100644 index 0000000..022115a --- /dev/null +++ b/papers/items/2026-2606-13994-hidden-in-plain-sight-benchmarking-agent-safety-against-decomposition-attacks-wi.md @@ -0,0 +1,64 @@ +# Paper: Hidden in Plain Sight: Benchmarking Agent Safety Against Decomposition Attacks with DECOMPBENCH + +--- +type: paper +title: "Hidden in Plain Sight: Benchmarking Agent Safety Against Decomposition Attacks with DECOMPBENCH" +authors: Vikhyath Kothamasu, Virginia Smith, Chhavi Yadav +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.13994 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-12 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: agent-safety, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, tool-use +- inferred topics: agent-evaluation, agent-safety, planning, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI, cs.LG +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.13994 diff --git a/papers/items/2026-2606-14106-naive-visual-memory-is-not-enough-a-failure-mode-study-of-gui-agents.md b/papers/items/2026-2606-14106-naive-visual-memory-is-not-enough-a-failure-mode-study-of-gui-agents.md new file mode 100644 index 0000000..869f5b7 --- /dev/null +++ b/papers/items/2026-2606-14106-naive-visual-memory-is-not-enough-a-failure-mode-study-of-gui-agents.md @@ -0,0 +1,63 @@ +# Paper: Naive Visual Memory is Not Enough: A Failure-Mode Study of GUI Agents + +--- +type: paper +title: "Naive Visual Memory is Not Enough: A Failure-Mode Study of GUI Agents" +authors: Seoyoung Choi, Minseok Ko, Hyunseok Lee, Kunwoong Kim, Woomin Song, Chanseok Jeon, Jinwoo Shin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.14106 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-12 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - computer-use + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: computer-use, memory, rag, reasoning, tool-use +- arXiv categories: cs.MA, cs.CV +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.14106 diff --git a/papers/items/2026-2606-14470-gitofthoughts-version-controlled-reasoning-and-agent-memory-you-can-replay-diff-.md b/papers/items/2026-2606-14470-gitofthoughts-version-controlled-reasoning-and-agent-memory-you-can-replay-diff-.md new file mode 100644 index 0000000..0cf043b --- /dev/null +++ b/papers/items/2026-2606-14470-gitofthoughts-version-controlled-reasoning-and-agent-memory-you-can-replay-diff-.md @@ -0,0 +1,64 @@ +# Paper: GitOfThoughts: Version-Controlled Reasoning and Agent Memory You Can Replay, Diff, and Merge + +--- +type: paper +title: "GitOfThoughts: Version-Controlled Reasoning and Agent Memory You Can Replay, Diff, and Merge" +authors: Pavan C Shekar, Abhishek H S, Aswanth Krishnan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.14470 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-12 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, coding-agent, memory, rag, reasoning +- arXiv categories: cs.AI, cs.CL, cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.14470 diff --git a/papers/items/2026-2606-14502-from-chatbot-to-digital-colleague-the-paradigm-shift-toward-persistent-autonomou.md b/papers/items/2026-2606-14502-from-chatbot-to-digital-colleague-the-paradigm-shift-toward-persistent-autonomou.md new file mode 100644 index 0000000..d4d3ea3 --- /dev/null +++ b/papers/items/2026-2606-14502-from-chatbot-to-digital-colleague-the-paradigm-shift-toward-persistent-autonomou.md @@ -0,0 +1,62 @@ +# Paper: From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI + +--- +type: paper +title: "From Chatbot to Digital Colleague: The Paradigm Shift Toward Persistent Autonomous AI" +authors: Yongheng Zhang, Ziang Liu, Jiaxuan Zhu, Shuai Wang, Xiangqi Chen, Haojing Huang, Jiayi Kuang, Siyu Chen, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.14502 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-12 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.14502 diff --git a/papers/items/2026-2606-14517-from-shield-to-target-denial-of-service-attacks-on-llm-based-agent-guardrails.md b/papers/items/2026-2606-14517-from-shield-to-target-denial-of-service-attacks-on-llm-based-agent-guardrails.md new file mode 100644 index 0000000..0ed47d1 --- /dev/null +++ b/papers/items/2026-2606-14517-from-shield-to-target-denial-of-service-attacks-on-llm-based-agent-guardrails.md @@ -0,0 +1,63 @@ +# Paper: From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails + +--- +type: paper +title: "From Shield to Target: Denial-of-Service Attacks on LLM-Based Agent Guardrails" +authors: Yuguang Zhou, Xunguang Wang, Pingchuan Ma, Zhantong Xue, Zhaoyu Wang, Shuai Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.14517 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-12 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation, autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, reasoning +- arXiv categories: cs.CR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.14517 diff --git a/papers/items/2026-2606-14571-streammembench-streaming-evaluation-of-agent-memory-for-future-oriented-assistan.md b/papers/items/2026-2606-14571-streammembench-streaming-evaluation-of-agent-memory-for-future-oriented-assistan.md new file mode 100644 index 0000000..8d19281 --- /dev/null +++ b/papers/items/2026-2606-14571-streammembench-streaming-evaluation-of-agent-memory-for-future-oriented-assistan.md @@ -0,0 +1,60 @@ +# Paper: StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance + +--- +type: paper +title: "StreamMemBench: Streaming Evaluation of Agent Memory for Future-Oriented Assistance" +authors: Guanming Liu, Yuqi Ren, Hansu Gu, Peng Zhang, Weihang Wang, Jiahao Liu, Ning Gu, Tun Lu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.14571 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-12 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.14571 diff --git a/papers/items/2026-2606-14574-simmer-benchmarking-latent-failures-in-llm-executable-planning-with-a-world-mode.md b/papers/items/2026-2606-14574-simmer-benchmarking-latent-failures-in-llm-executable-planning-with-a-world-mode.md new file mode 100644 index 0000000..8e1ec38 --- /dev/null +++ b/papers/items/2026-2606-14574-simmer-benchmarking-latent-failures-in-llm-executable-planning-with-a-world-mode.md @@ -0,0 +1,64 @@ +# Paper: SIMMER: Benchmarking Latent Failures in LLM Executable Planning with a World Model + +--- +type: paper +title: "SIMMER: Benchmarking Latent Failures in LLM Executable Planning with a World Model" +authors: Xiaoxin Lu, Ranran Haoran Zhang, Rui Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.14574 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-12 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use, world-model +- arXiv categories: cs.CL, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.14574 diff --git a/papers/items/2026-2606-14790-xflow-an-executable-protocol-programming-system-for-reliable-multi-agent-workflo.md b/papers/items/2026-2606-14790-xflow-an-executable-protocol-programming-system-for-reliable-multi-agent-workflo.md new file mode 100644 index 0000000..bb51728 --- /dev/null +++ b/papers/items/2026-2606-14790-xflow-an-executable-protocol-programming-system-for-reliable-multi-agent-workflo.md @@ -0,0 +1,65 @@ +# Paper: XFlow: An Executable Protocol Programming System for Reliable Multi-Agent Workflows + +--- +type: paper +title: "XFlow: An Executable Protocol Programming System for Reliable Multi-Agent Workflows" +authors: Hanqi Li, Jing Peng, Zijian Wang, Lu Chen, Kai Yu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.14790 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - coding-agent + - memory + - multi-agent + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.PL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: coding-agent, memory, multi-agent, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.PL, cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.14790 diff --git a/papers/items/2026-2606-14805-knowledge-based-zero-replay-debugging-of-multi-agent-llm-traces.md b/papers/items/2026-2606-14805-knowledge-based-zero-replay-debugging-of-multi-agent-llm-traces.md new file mode 100644 index 0000000..de3fdc7 --- /dev/null +++ b/papers/items/2026-2606-14805-knowledge-based-zero-replay-debugging-of-multi-agent-llm-traces.md @@ -0,0 +1,61 @@ +# Paper: Knowledge-Based Zero-Replay Debugging of Multi-Agent LLM Traces + +--- +type: paper +title: Knowledge-Based Zero-Replay Debugging of Multi-Agent LLM Traces +authors: Dong Ho Kang, Hyeonjeong Cha, Daein Weon +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.14805 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - memory + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: memory, multi-agent, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.14805 diff --git a/papers/items/2026-2606-15017-are-online-skill-and-memory-modules-always-worth-their-tokens-a-budget-constrain.md b/papers/items/2026-2606-15017-are-online-skill-and-memory-modules-always-worth-their-tokens-a-budget-constrain.md new file mode 100644 index 0000000..5910761 --- /dev/null +++ b/papers/items/2026-2606-15017-are-online-skill-and-memory-modules-always-worth-their-tokens-a-budget-constrain.md @@ -0,0 +1,62 @@ +# Paper: Are Online Skill and Memory Modules Always Worth Their Tokens? A Budget-Constrained Study of Web Agents + +--- +type: paper +title: Are Online Skill and Memory Modules Always Worth Their Tokens? A Budget-Constrained Study of Web Agents +authors: Sina Hajimiri, Masih Aminbeidokhti, Jose Dolz, Ismail Ben Ayed, Issam H. Laradji, Spandana Gella, Nicolas Gontier +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15017 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-12 +updated_at: 2026-06-12 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, memory, reasoning, workflow-agent +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15017 diff --git a/papers/items/2026-2606-15034-osguard-a-benchmark-for-safety-in-computer-use-agents.md b/papers/items/2026-2606-15034-osguard-a-benchmark-for-safety-in-computer-use-agents.md new file mode 100644 index 0000000..22b31ed --- /dev/null +++ b/papers/items/2026-2606-15034-osguard-a-benchmark-for-safety-in-computer-use-agents.md @@ -0,0 +1,61 @@ +# Paper: OSGuard: A Benchmark for Safety in Computer-Use Agents + +--- +type: paper +title: "OSGuard: A Benchmark for Safety in Computer-Use Agents" +authors: Mina Mohammadmirzaei, Jeffrey Flanigan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15034 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-13 +updated_at: 2026-06-13 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15034 diff --git a/papers/items/2026-2606-15079-ling-and-ring-2-6-technical-report-efficient-and-instant-agentic-intelligence-at.md b/papers/items/2026-2606-15079-ling-and-ring-2-6-technical-report-efficient-and-instant-agentic-intelligence-at.md new file mode 100644 index 0000000..1ebf1d2 --- /dev/null +++ b/papers/items/2026-2606-15079-ling-and-ring-2-6-technical-report-efficient-and-instant-agentic-intelligence-at.md @@ -0,0 +1,64 @@ +# Paper: Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale + +--- +type: paper +title: "Ling and Ring 2.6 Technical Report: Efficient and Instant Agentic Intelligence at Trillion-Parameter Scale" +authors: Ang Li, Ben Liu, Bin Han, Bin Hu, Bin Jing, Binbin Hu, Bing Li, Cai Chen, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15079 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-13 +updated_at: 2026-06-13 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - computer-use + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-safety, coding-agent, computer-use, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CL, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15079 diff --git a/papers/items/2026-2606-15152-can-agents-read-the-room-benchmarking-visual-social-intelligence-in-multimodal-s.md b/papers/items/2026-2606-15152-can-agents-read-the-room-benchmarking-visual-social-intelligence-in-multimodal-s.md new file mode 100644 index 0000000..adfa810 --- /dev/null +++ b/papers/items/2026-2606-15152-can-agents-read-the-room-benchmarking-visual-social-intelligence-in-multimodal-s.md @@ -0,0 +1,61 @@ +# Paper: Can Agents Read the Room? Benchmarking Visual Social Intelligence in Multimodal Simulation + +--- +type: paper +title: Can Agents Read the Room? Benchmarking Visual Social Intelligence in Multimodal Simulation +authors: Shijun Wan, Xuehai Wu, Jiwen Zhang, Siyuan Wang, Zhongyu Wei +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15152 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-13 +updated_at: 2026-06-13 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, computer-use, tool-use, world-model +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15152 diff --git a/papers/items/2026-2606-15242-benign-in-isolation-harmful-in-composition-security-risks-in-agent-skill-ecosyst.md b/papers/items/2026-2606-15242-benign-in-isolation-harmful-in-composition-security-risks-in-agent-skill-ecosyst.md new file mode 100644 index 0000000..fc1180a --- /dev/null +++ b/papers/items/2026-2606-15242-benign-in-isolation-harmful-in-composition-security-risks-in-agent-skill-ecosyst.md @@ -0,0 +1,62 @@ +# Paper: Benign in Isolation, Harmful in Composition: Security Risks in Agent Skill Ecosystems + +--- +type: paper +title: "Benign in Isolation, Harmful in Composition: Security Risks in Agent Skill Ecosystems" +authors: Yi Xie, Jiawei Du, Yu Cheng, Jiuan Zhou, Zhaoxia Yin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15242 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-13 +updated_at: 2026-06-13 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15242 diff --git a/papers/items/2026-2606-15376-coagent-concurrency-control-for-multi-agent-systems.md b/papers/items/2026-2606-15376-coagent-concurrency-control-for-multi-agent-systems.md new file mode 100644 index 0000000..ddaf47e --- /dev/null +++ b/papers/items/2026-2606-15376-coagent-concurrency-control-for-multi-agent-systems.md @@ -0,0 +1,63 @@ +# Paper: CoAgent: Concurrency Control for Multi-Agent Systems + +--- +type: paper +title: "CoAgent: Concurrency Control for Multi-Agent Systems" +authors: Hongtao Lyu, Dingyan Zhang, Mingyu Wu, Xingda Wei, Haibo Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15376 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-13 +updated_at: 2026-06-13 +status: queued +relevance: high +topics: + - coding-agent + - multi-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DC + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: coding-agent, multi-agent, planning, tool-use +- arXiv categories: cs.DC, cs.AI, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15376 diff --git a/papers/items/2026-2606-15591-agentic-retrieval-and-reinforcement-learned-equation-chains-a-controlled-generat.md b/papers/items/2026-2606-15591-agentic-retrieval-and-reinforcement-learned-equation-chains-a-controlled-generat.md new file mode 100644 index 0000000..8aabbcb --- /dev/null +++ b/papers/items/2026-2606-15591-agentic-retrieval-and-reinforcement-learned-equation-chains-a-controlled-generat.md @@ -0,0 +1,62 @@ +# Paper: Agentic Retrieval and Reinforcement Learned Equation Chains: A Controlled Generation Framework for Complex and Novel Physics Word Problems + +--- +type: paper +title: "Agentic Retrieval and Reinforcement Learned Equation Chains: A Controlled Generation Framework for Complex and Novel Physics Word Problems" +authors: Tirthankar Mittra +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15591 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-14 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, computer-use, rag +- arXiv categories: cs.AI, cs.CL, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15591 diff --git a/papers/items/2026-2606-15609-fragfuse-bypassing-access-control-of-large-language-model-agents-via-memory-base.md b/papers/items/2026-2606-15609-fragfuse-bypassing-access-control-of-large-language-model-agents-via-memory-base.md new file mode 100644 index 0000000..e8688cc --- /dev/null +++ b/papers/items/2026-2606-15609-fragfuse-bypassing-access-control-of-large-language-model-agents-via-memory-base.md @@ -0,0 +1,62 @@ +# Paper: FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion + +--- +type: paper +title: "FragFuse: Bypassing Access Control of Large Language Model Agents via Memory-Based Query Fragmentation and Fusion" +authors: Zixin Rao, Wentian Zhu, Chan Aristella Lu, Zhaorun Chen, Wei Niu, Le Guan, Bo Li, Zhen Xiang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15609 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-14 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15609 diff --git a/papers/items/2026-2606-15684-multi-agent-framework-for-time-sensitive-complementary-collaboration-in-minecraf.md b/papers/items/2026-2606-15684-multi-agent-framework-for-time-sensitive-complementary-collaboration-in-minecraf.md new file mode 100644 index 0000000..8178d58 --- /dev/null +++ b/papers/items/2026-2606-15684-multi-agent-framework-for-time-sensitive-complementary-collaboration-in-minecraf.md @@ -0,0 +1,62 @@ +# Paper: Multi-agent Framework for Time-Sensitive Complementary Collaboration in Minecraft + +--- +type: paper +title: Multi-agent Framework for Time-Sensitive Complementary Collaboration in Minecraft +authors: Juheon Yi, Jinglu Wang, Xiaoyi Zhang, Yan Lu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15684 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-14 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15684 diff --git a/papers/items/2026-2606-15709-ai-driven-framework-for-adaptive-water-network-management-with-proof-of-concept-.md b/papers/items/2026-2606-15709-ai-driven-framework-for-adaptive-water-network-management-with-proof-of-concept-.md new file mode 100644 index 0000000..052648f --- /dev/null +++ b/papers/items/2026-2606-15709-ai-driven-framework-for-adaptive-water-network-management-with-proof-of-concept-.md @@ -0,0 +1,64 @@ +# Paper: AI-Driven Framework for Adaptive Water Network Management with Proof-of-Concept Implementation: Addressing Non-Revenue Water in Jordan + +--- +type: paper +title: "AI-Driven Framework for Adaptive Water Network Management with Proof-of-Concept Implementation: Addressing Non-Revenue Water in Jordan" +authors: Mohammed Fasha, Nahel Al-Maayta, Bilal Sowan, Mohammad Athamneh, Husam Barham +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15709 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-14 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: function-calling, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling, rag-agent +- inferred topics: agent-evaluation, agent-safety, rag, tool-use, workflow-agent, world-model +- arXiv categories: cs.AI, cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15709 diff --git a/papers/items/2026-2606-15862-retailbench-benchmarking-long-horizon-reasoning-and-coherent-decision-making-of-.md b/papers/items/2026-2606-15862-retailbench-benchmarking-long-horizon-reasoning-and-coherent-decision-making-of-.md new file mode 100644 index 0000000..f729ca4 --- /dev/null +++ b/papers/items/2026-2606-15862-retailbench-benchmarking-long-horizon-reasoning-and-coherent-decision-making-of-.md @@ -0,0 +1,62 @@ +# Paper: RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments + +--- +type: paper +title: "RetailBench: Benchmarking long horizon reasoning and coherent decision making of LLM agents in realistic retail environments" +authors: Linghua Zhang, Jun Wang, Jingtong Wu, Zhisong Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15862 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, planning, reasoning, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15862 diff --git a/papers/items/2026-2606-15874-llm-as-code-agentic-programming-for-agent-harness.md b/papers/items/2026-2606-15874-llm-as-code-agentic-programming-for-agent-harness.md new file mode 100644 index 0000000..034458c --- /dev/null +++ b/papers/items/2026-2606-15874-llm-as-code-agentic-programming-for-agent-harness.md @@ -0,0 +1,60 @@ +# Paper: LLM-as-Code: Agentic Programming for Agent Harness + +--- +type: paper +title: "LLM-as-Code: Agentic Programming for Agent Harness" +authors: Junjia Qi, Zichuan Fu, Jingtong Gao, Wenlin Zhang, Hanyu Yan, Xian Wu, Xiangyu Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15874 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: reasoning, tool-use +- arXiv categories: cs.AI, cs.SE +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15874 diff --git a/papers/items/2026-2606-15903-control-plane-placement-shapes-forgetting-an-architectural-study-of-agent-memory.md b/papers/items/2026-2606-15903-control-plane-placement-shapes-forgetting-an-architectural-study-of-agent-memory.md new file mode 100644 index 0000000..6df9d76 --- /dev/null +++ b/papers/items/2026-2606-15903-control-plane-placement-shapes-forgetting-an-architectural-study-of-agent-memory.md @@ -0,0 +1,62 @@ +# Paper: Control-Plane Placement Shapes Forgetting: An Architectural Study of Agent Memory Across Thirteen System Configurations + +--- +type: paper +title: "Control-Plane Placement Shapes Forgetting: An Architectural Study of Agent Memory Across Thirteen System Configurations" +authors: Dongxu Yang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15903 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag +- arXiv categories: cs.CL, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15903 diff --git a/papers/items/2026-2606-15906-mage-rag-multigranular-adaptive-graph-evidence-for-agentic-multimodal-rag-in-lon.md b/papers/items/2026-2606-15906-mage-rag-multigranular-adaptive-graph-evidence-for-agentic-multimodal-rag-in-lon.md new file mode 100644 index 0000000..6ff01d1 --- /dev/null +++ b/papers/items/2026-2606-15906-mage-rag-multigranular-adaptive-graph-evidence-for-agentic-multimodal-rag-in-lon.md @@ -0,0 +1,64 @@ +# Paper: MAGE-RAG: Multigranular Adaptive Graph Evidence for Agentic Multimodal RAG in Long-Document QA + +--- +type: paper +title: "MAGE-RAG: Multigranular Adaptive Graph Evidence for Agentic Multimodal RAG in Long-Document QA" +authors: Yilong Zuo, Xunkai Li, Jing Yuan, Qiangqiang Dai, Hongchao Qin, Ronghua Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15906 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-14 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI + - cs.CL + - cs.DB + - cs.MM +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, coding-agent, rag +- arXiv categories: cs.IR, cs.AI, cs.CL, cs.DB, cs.MM +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15906 diff --git a/papers/items/2026-2606-15931-deeproot-a-kg-coordinated-multi-agent-system-for-therapeutic-reasoning-over-hist.md b/papers/items/2026-2606-15931-deeproot-a-kg-coordinated-multi-agent-system-for-therapeutic-reasoning-over-hist.md new file mode 100644 index 0000000..2a3a612 --- /dev/null +++ b/papers/items/2026-2606-15931-deeproot-a-kg-coordinated-multi-agent-system-for-therapeutic-reasoning-over-hist.md @@ -0,0 +1,63 @@ +# Paper: DeepRoot: A KG-Coordinated Multi-Agent System for Therapeutic Reasoning over Historical Medical Texts + +--- +type: paper +title: "DeepRoot: A KG-Coordinated Multi-Agent System for Therapeutic Reasoning over Historical Medical Texts" +authors: Zijian Carl Ma, Sean J. Wang, Sijbren Kramer, Li Erran Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15931 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-14 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.MA, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15931 diff --git a/papers/items/2026-2606-15994-agentic-framework-for-deep-learning-workload-migration-via-in-context-learning.md b/papers/items/2026-2606-15994-agentic-framework-for-deep-learning-workload-migration-via-in-context-learning.md new file mode 100644 index 0000000..08cf8f6 --- /dev/null +++ b/papers/items/2026-2606-15994-agentic-framework-for-deep-learning-workload-migration-via-in-context-learning.md @@ -0,0 +1,61 @@ +# Paper: Agentic Framework for Deep Learning workload migration via In-Context Learning + +--- +type: paper +title: Agentic Framework for Deep Learning workload migration via In-Context Learning +authors: Qiyue Liang, Steven Ingram, George Vanica, Andi Gavrilescu, Newfel Harrat, Hassan Sipra, Sethuraman Sankaran +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.15994 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-14 +updated_at: 2026-06-14 +status: queued +relevance: high +topics: + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-safety, rag, tool-use +- arXiv categories: cs.AI, cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.15994 diff --git a/papers/items/2026-2606-16111-towards-pareto-optimal-tool-integrated-agents-with-pareto-ranking-policy-optimiz.md b/papers/items/2026-2606-16111-towards-pareto-optimal-tool-integrated-agents-with-pareto-ranking-policy-optimiz.md new file mode 100644 index 0000000..783849f --- /dev/null +++ b/papers/items/2026-2606-16111-towards-pareto-optimal-tool-integrated-agents-with-pareto-ranking-policy-optimiz.md @@ -0,0 +1,62 @@ +# Paper: Towards Pareto-Optimal Tool-Integrated Agents with Pareto Ranking Policy Optimization + +--- +type: paper +title: Towards Pareto-Optimal Tool-Integrated Agents with Pareto Ranking Policy Optimization +authors: Junyi Li, Xiaowei Qian, Yingyi Zhang, Wenlin Zhang, Guojing Li, Sheng Zhang, Xiao Han, Yichao Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16111 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent, tool-use +- inferred topics: agent-safety, computer-use, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16111 diff --git a/papers/items/2026-2606-16295-visualclaw-a-real-time-personalized-agent-for-the-physical-world.md b/papers/items/2026-2606-16295-visualclaw-a-real-time-personalized-agent-for-the-physical-world.md new file mode 100644 index 0000000..3fb23b9 --- /dev/null +++ b/papers/items/2026-2606-16295-visualclaw-a-real-time-personalized-agent-for-the-physical-world.md @@ -0,0 +1,63 @@ +# Paper: VisualClaw: A Real-Time, Personalized Agent for the Physical World + +--- +type: paper +title: "VisualClaw: A Real-Time, Personalized Agent for the Physical World" +authors: Haoqin Tu, Jianwen Chen, Zijun Wang, Siwei Han, Juncheng Wu, Hardy Chen, Haonian Ji, Kaiwen Xiong, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16295 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation, tool-use, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, tool-use, web-gui-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, rag, tool-use +- arXiv categories: cs.CV, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16295 diff --git a/papers/items/2026-2606-16420-transferable-self-evolving-playbooks-for-agentic-security-auditing.md b/papers/items/2026-2606-16420-transferable-self-evolving-playbooks-for-agentic-security-auditing.md new file mode 100644 index 0000000..214fa44 --- /dev/null +++ b/papers/items/2026-2606-16420-transferable-self-evolving-playbooks-for-agentic-security-auditing.md @@ -0,0 +1,64 @@ +# Paper: Transferable Self-Evolving Playbooks for Agentic Security Auditing + +--- +type: paper +title: Transferable Self-Evolving Playbooks for Agentic Security Auditing +authors: Ziyue Wang, Cheuk Wang Maurice Ng, Chenchen Yu, Strick Sheng, Kaihua Qin, Liyi Zhou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16420 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - embodied-agent + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-safety, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, tool-use +- inferred topics: agent-evaluation, agent-safety, computer-use, embodied-agent, rag, tool-use, workflow-agent +- arXiv categories: cs.CR +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16420 diff --git a/papers/items/2026-2606-16432-accord-action-conditioned-contextual-grounding-for-language-agents.md b/papers/items/2026-2606-16432-accord-action-conditioned-contextual-grounding-for-language-agents.md new file mode 100644 index 0000000..e592074 --- /dev/null +++ b/papers/items/2026-2606-16432-accord-action-conditioned-contextual-grounding-for-language-agents.md @@ -0,0 +1,62 @@ +# Paper: ACCORD: Action-Conditioned Contextual Grounding for Language Agents + +--- +type: paper +title: "ACCORD: Action-Conditioned Contextual Grounding for Language Agents" +authors: Lai Jiang, Cheng Qian, Zhenhailong Wang, Pan Lu, Heng Ji, Hao Peng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16432 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, embodied-agent, rag, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16432 diff --git a/papers/items/2026-2606-16481-steering-emotional-dynamics-for-art-therapy-controllable-narrative-script-genera.md b/papers/items/2026-2606-16481-steering-emotional-dynamics-for-art-therapy-controllable-narrative-script-genera.md new file mode 100644 index 0000000..ac9b919 --- /dev/null +++ b/papers/items/2026-2606-16481-steering-emotional-dynamics-for-art-therapy-controllable-narrative-script-genera.md @@ -0,0 +1,60 @@ +# Paper: Steering Emotional Dynamics for Art Therapy: Controllable Narrative Script Generation through Hierarchically Guided LLM Agents + +--- +type: paper +title: "Steering Emotional Dynamics for Art Therapy: Controllable Narrative Script Generation through Hierarchically Guided LLM Agents" +authors: Suqing Wang, Qinghai Miao, Chao Guo, Yisheng Lv +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16481 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: computer-use, planning, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16481 diff --git a/papers/items/2026-2606-16534-generated-parallel-scalable-a-study-of-agentic-ai-generated-julia-code-on-superc.md b/papers/items/2026-2606-16534-generated-parallel-scalable-a-study-of-agentic-ai-generated-julia-code-on-superc.md new file mode 100644 index 0000000..260bc4a --- /dev/null +++ b/papers/items/2026-2606-16534-generated-parallel-scalable-a-study-of-agentic-ai-generated-julia-code-on-superc.md @@ -0,0 +1,61 @@ +# Paper: Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers + +--- +type: paper +title: Generated, Parallel, Scalable? A Study of Agentic AI-Generated Julia Code on Supercomputers +authors: Linus Bantel, Anna-Lena Roth, Jonas Posner, Dirk Pflüger +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16534 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, memory, planning, tool-use +- arXiv categories: cs.DC +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16534 diff --git a/papers/items/2026-2606-16576-can-llm-agents-infer-world-models-evidence-from-agentic-automata-learning.md b/papers/items/2026-2606-16576-can-llm-agents-infer-world-models-evidence-from-agentic-automata-learning.md new file mode 100644 index 0000000..49875d4 --- /dev/null +++ b/papers/items/2026-2606-16576-can-llm-agents-infer-world-models-evidence-from-agentic-automata-learning.md @@ -0,0 +1,62 @@ +# Paper: Can LLM Agents Infer World Models? Evidence from Agentic Automata Learning + +--- +type: paper +title: Can LLM Agents Infer World Models? Evidence from Agentic Automata Learning +authors: Reef Menaged, Gili Lior, Shauli Ravfogel, Roee Aharoni, Gabriel Stanovsky +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16576 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, reasoning, tool-use, world-model +- arXiv categories: cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16576 diff --git a/papers/items/2026-2606-16591-sing-synthetic-intention-graph-for-scalable-active-tool-discovery-in-llm-agents.md b/papers/items/2026-2606-16591-sing-synthetic-intention-graph-for-scalable-active-tool-discovery-in-llm-agents.md new file mode 100644 index 0000000..cfad235 --- /dev/null +++ b/papers/items/2026-2606-16591-sing-synthetic-intention-graph-for-scalable-active-tool-discovery-in-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: SING: Synthetic Intention Graph for Scalable Active Tool Discovery in LLM Agents + +--- +type: paper +title: "SING: Synthetic Intention Graph for Scalable Active Tool Discovery in LLM Agents" +authors: Qiao Xiao, Haochen Shi, Yisen Gao, Wenbin Hu, Huihao Jing, Tianshi Zheng, Baixuan Xu, Ziheng Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16591 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, multi-agent, planning, rag, tool-use +- arXiv categories: cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16591 diff --git a/papers/items/2026-2606-16613-coffeebench-benchmarking-long-horizon-llm-agents-in-heterogeneous-multi-agent-ec.md b/papers/items/2026-2606-16613-coffeebench-benchmarking-long-horizon-llm-agents-in-heterogeneous-multi-agent-ec.md new file mode 100644 index 0000000..89ecf85 --- /dev/null +++ b/papers/items/2026-2606-16613-coffeebench-benchmarking-long-horizon-llm-agents-in-heterogeneous-multi-agent-ec.md @@ -0,0 +1,62 @@ +# Paper: CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies + +--- +type: paper +title: "CoffeeBench: Benchmarking Long-Horizon LLM Agents in Heterogeneous Multi-Agent Economies" +authors: Issa Sugiura, Daichi Hattori, Kazuo Araragi, Keita Ogawa, Shota Onose, Taro Makino, Teppei Usuki, Takashi Ishida +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16613 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: autonomous-agent-llm, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, planning-agent +- inferred topics: agent-evaluation, multi-agent, planning, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16613 diff --git a/papers/items/2026-2606-16659-fraudsmswalker-benchmarking-agentic-large-language-models-for-sms-to-webpage-fra.md b/papers/items/2026-2606-16659-fraudsmswalker-benchmarking-agentic-large-language-models-for-sms-to-webpage-fra.md new file mode 100644 index 0000000..1d321fa --- /dev/null +++ b/papers/items/2026-2606-16659-fraudsmswalker-benchmarking-agentic-large-language-models-for-sms-to-webpage-fra.md @@ -0,0 +1,61 @@ +# Paper: FraudSMSWalker: Benchmarking Agentic Large Language Models for SMS-to-Webpage Fraud Detection + +--- +type: paper +title: "FraudSMSWalker: Benchmarking Agentic Large Language Models for SMS-to-Webpage Fraud Detection" +authors: Y. H. Zhou, Z. M. Ma, Y. J. Zhou, Y. T. Li, H. X. Xiang, Y. M. Cheng, T. L. Chen, K. J. Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16659 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, tool-use +- arXiv categories: cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16659 diff --git a/papers/items/2026-2606-16748-mypcbench-a-benchmark-for-personally-intelligent-computer-use-agents.md b/papers/items/2026-2606-16748-mypcbench-a-benchmark-for-personally-intelligent-computer-use-agents.md new file mode 100644 index 0000000..b926176 --- /dev/null +++ b/papers/items/2026-2606-16748-mypcbench-a-benchmark-for-personally-intelligent-computer-use-agents.md @@ -0,0 +1,62 @@ +# Paper: MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents + +--- +type: paper +title: "MyPCBench: A Benchmark for Personally Intelligent Computer-Use Agents" +authors: Lawrence Keunho Jang, Andrew Keunwoo Jang, Jing Yu Koh, Ruslan Salakhutdinov +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16748 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, web-gui-agent +- inferred topics: agent-evaluation, computer-use, tool-use, workflow-agent +- arXiv categories: cs.LG, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16748 diff --git a/papers/items/2026-2606-16774-openclaw-skill-collective-skill-tree-search-for-agentic-large-language-models.md b/papers/items/2026-2606-16774-openclaw-skill-collective-skill-tree-search-for-agentic-large-language-models.md new file mode 100644 index 0000000..9555944 --- /dev/null +++ b/papers/items/2026-2606-16774-openclaw-skill-collective-skill-tree-search-for-agentic-large-language-models.md @@ -0,0 +1,63 @@ +# Paper: OpenClaw-Skill: Collective Skill Tree Search for Agentic Large Language Models + +--- +type: paper +title: "OpenClaw-Skill: Collective Skill Tree Search for Agentic Large Language Models" +authors: Tianyi Lin, Chuanyu Sun, Jingyi Zhang, Changxu Wei, Huanjin Yao, Shunyu Liu, Xikun Zhang, Liu Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16774 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: planning-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent, tool-use +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16774 diff --git a/papers/items/2026-2606-16802-labosbench-benchmarking-computer-use-agents-for-scientific-instrument-control.md b/papers/items/2026-2606-16802-labosbench-benchmarking-computer-use-agents-for-scientific-instrument-control.md new file mode 100644 index 0000000..45b208e --- /dev/null +++ b/papers/items/2026-2606-16802-labosbench-benchmarking-computer-use-agents-for-scientific-instrument-control.md @@ -0,0 +1,62 @@ +# Paper: LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control + +--- +type: paper +title: "LabOSBench: Benchmarking Computer Use Agents for Scientific Instrument Control" +authors: Anqi Zou, Han Deng, Chengyu Zhang, Junquan Hu, Yu Wang, Yuxiang Xing, Aokai Zhang, Hanling Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16802 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - planning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, planning, workflow-agent +- arXiv categories: cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16802 diff --git a/papers/items/2026-2606-16813-gist-cmtf-goal-state-inference-for-causal-minimal-tool-filtering-in-llm-agents.md b/papers/items/2026-2606-16813-gist-cmtf-goal-state-inference-for-causal-minimal-tool-filtering-in-llm-agents.md new file mode 100644 index 0000000..ae83c32 --- /dev/null +++ b/papers/items/2026-2606-16813-gist-cmtf-goal-state-inference-for-causal-minimal-tool-filtering-in-llm-agents.md @@ -0,0 +1,60 @@ +# Paper: GIST-CMTF: Goal-State Inference for Causal Minimal Tool Filtering in LLM Agents + +--- +type: paper +title: "GIST-CMTF: Goal-State Inference for Causal Minimal Tool Filtering in LLM Agents" +authors: Rahul Suresh Babu, Rohit Shukla +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16813 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, computer-use, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16813 diff --git a/papers/items/2026-2606-16839-towards-llm-accelerated-rapid-reviews-for-software-tool-discovery-case-for-log-a.md b/papers/items/2026-2606-16839-towards-llm-accelerated-rapid-reviews-for-software-tool-discovery-case-for-log-a.md new file mode 100644 index 0000000..cdc3916 --- /dev/null +++ b/papers/items/2026-2606-16839-towards-llm-accelerated-rapid-reviews-for-software-tool-discovery-case-for-log-a.md @@ -0,0 +1,62 @@ +# Paper: Towards LLM Accelerated Rapid Reviews for Software Tool Discovery -- Case for Log Anomaly Detection + +--- +type: paper +title: Towards LLM Accelerated Rapid Reviews for Software Tool Discovery -- Case for Log Anomaly Detection +authors: Jesse Nyyssölä, Hamza Bin Mazhar, Alexander Bakhtin, Matteo Esposito, Nana Reinikainen, Yuqing Wang, Ying Song, Davide Taibi, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16839 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, coding-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.SE +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16839 diff --git a/papers/items/2026-2606-16871-human-on-the-bridge-scalable-evaluation-for-ai-agents.md b/papers/items/2026-2606-16871-human-on-the-bridge-scalable-evaluation-for-ai-agents.md new file mode 100644 index 0000000..4a9ff10 --- /dev/null +++ b/papers/items/2026-2606-16871-human-on-the-bridge-scalable-evaluation-for-ai-agents.md @@ -0,0 +1,62 @@ +# Paper: Human-on-the-Bridge: Scalable Evaluation for AI Agents + +--- +type: paper +title: "Human-on-the-Bridge: Scalable Evaluation for AI Agents" +authors: Fouad Bousetouane +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.16871 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, computer-use, memory, rag, tool-use +- arXiv categories: cs.MA +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.16871 diff --git a/papers/items/2026-2606-17041-benchmarking-llm-agents-on-meta-analysis-articles-from-nature-portfolio.md b/papers/items/2026-2606-17041-benchmarking-llm-agents-on-meta-analysis-articles-from-nature-portfolio.md new file mode 100644 index 0000000..e193ac3 --- /dev/null +++ b/papers/items/2026-2606-17041-benchmarking-llm-agents-on-meta-analysis-articles-from-nature-portfolio.md @@ -0,0 +1,63 @@ +# Paper: Benchmarking LLM Agents on Meta-Analysis Articles from Nature Portfolio + +--- +type: paper +title: Benchmarking LLM Agents on Meta-Analysis Articles from Nature Portfolio +authors: Anzhe Xie, Weihang Su, Yujia Zhou, Yiqun Liu, Qingyao Ai +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17041 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, computer-use, rag, reasoning, workflow-agent +- arXiv categories: cs.CL, cs.IR +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17041 diff --git a/papers/items/2026-2606-17076-cmip-forge-an-agentic-system-that-retrieves-computes-and-self-reviews-climate-sc.md b/papers/items/2026-2606-17076-cmip-forge-an-agentic-system-that-retrieves-computes-and-self-reviews-climate-sc.md new file mode 100644 index 0000000..88fab2c --- /dev/null +++ b/papers/items/2026-2606-17076-cmip-forge-an-agentic-system-that-retrieves-computes-and-self-reviews-climate-sc.md @@ -0,0 +1,64 @@ +# Paper: CMIP-Forge: An Agentic System that Retrieves, Computes, and Self-Reviews Climate Science + +--- +type: paper +title: "CMIP-Forge: An Agentic System that Retrieves, Computes, and Self-Reviews Climate Science" +authors: Dmitrii Pantiukhin, Boris Shapkin, Ivan Kuznetsov, Thomas Jung, Nikolay Koldunov +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17076 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-10 +updated_at: 2026-06-10 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - physics.ao-ph + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, agent-safety, planning, rag, tool-use, workflow-agent +- arXiv categories: physics.ao-ph, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17076 diff --git a/papers/items/2026-2606-17114-an-evaluation-of-data-leakage-risks-in-tool-using-llm-agents-in-realistic-scenar.md b/papers/items/2026-2606-17114-an-evaluation-of-data-leakage-risks-in-tool-using-llm-agents-in-realistic-scenar.md new file mode 100644 index 0000000..d34b019 --- /dev/null +++ b/papers/items/2026-2606-17114-an-evaluation-of-data-leakage-risks-in-tool-using-llm-agents-in-realistic-scenar.md @@ -0,0 +1,63 @@ +# Paper: An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios + +--- +type: paper +title: An Evaluation of Data Leakage Risks in Tool-Using LLM Agents in Realistic Scenarios +authors: Hankyul Baek, Jaewon Noh, Sang Seo, Yongsu Kim, Gabriel Waikin Loh Matienzo, Young Il Kim, Ee Wei Seah, Akriti Vij +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17114 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-safety, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, tool-use +- inferred topics: agent-evaluation, agent-safety, tool-use, workflow-agent, world-model +- arXiv categories: cs.CR, cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17114 diff --git a/papers/items/2026-2606-17246-geodisaster-benchmarking-orchestrated-agents-for-operational-disaster-geo-intell.md b/papers/items/2026-2606-17246-geodisaster-benchmarking-orchestrated-agents-for-operational-disaster-geo-intell.md new file mode 100644 index 0000000..f3a6f15 --- /dev/null +++ b/papers/items/2026-2606-17246-geodisaster-benchmarking-orchestrated-agents-for-operational-disaster-geo-intell.md @@ -0,0 +1,65 @@ +# Paper: GeoDisaster: Benchmarking Orchestrated Agents for Operational Disaster Geo-Intelligence + +--- +type: paper +title: "GeoDisaster: Benchmarking Orchestrated Agents for Operational Disaster Geo-Intelligence" +authors: Maram Hasan, Aman Verma, Savitra Roy, Hariseetharam Gunduboina, Daksh Jain, Muhammad Haris Khan, Subhasis Chaudhuri, Biplab Banerjee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17246 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CV, cs.MA +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17246 diff --git a/papers/items/2026-2606-17368-distributed-general-purpose-agent-networks-architecture-key-mechanisms-and-proto.md b/papers/items/2026-2606-17368-distributed-general-purpose-agent-networks-architecture-key-mechanisms-and-proto.md new file mode 100644 index 0000000..554197a --- /dev/null +++ b/papers/items/2026-2606-17368-distributed-general-purpose-agent-networks-architecture-key-mechanisms-and-proto.md @@ -0,0 +1,63 @@ +# Paper: Distributed General-Purpose Agent Networks: Architecture, Key Mechanisms, and Prototypes + +--- +type: paper +title: "Distributed General-Purpose Agent Networks: Architecture, Key Mechanisms, and Prototypes" +authors: Shengli Zhang, Deen Ma, Zibin Lin, Taotao Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17368 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-15 +updated_at: 2026-06-15 +status: queued +relevance: high +topics: + - computer-use + - multi-agent + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.NI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: computer-use, multi-agent, planning, tool-use, world-model +- arXiv categories: cs.AI, cs.NI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17368 diff --git a/papers/items/2026-2606-17383-model-validation-of-agentic-ai-systems-a-pomdp-based-framework-for-belief-state-.md b/papers/items/2026-2606-17383-model-validation-of-agentic-ai-systems-a-pomdp-based-framework-for-belief-state-.md new file mode 100644 index 0000000..78f4ff0 --- /dev/null +++ b/papers/items/2026-2606-17383-model-validation-of-agentic-ai-systems-a-pomdp-based-framework-for-belief-state-.md @@ -0,0 +1,63 @@ +# Paper: Model Validation of Agentic AI Systems: A POMDP-Based Framework for Belief-State, Forecast, and Policy Validation + +--- +type: paper +title: "Model Validation of Agentic AI Systems: A POMDP-Based Framework for Belief-State, Forecast, and Policy Validation" +authors: Matthew Francis Dixon +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17383 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - q-fin.RM + - cs.AI + - cs.LG + - stat.ML +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-safety, rag, tool-use +- arXiv categories: q-fin.RM, cs.AI, cs.LG, stat.ML +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17383 diff --git a/papers/items/2026-2606-17449-mode-rag-manifold-outlier-diagnosis-and-energy-based-retrieval-augmented-generat.md b/papers/items/2026-2606-17449-mode-rag-manifold-outlier-diagnosis-and-energy-based-retrieval-augmented-generat.md new file mode 100644 index 0000000..ef590d3 --- /dev/null +++ b/papers/items/2026-2606-17449-mode-rag-manifold-outlier-diagnosis-and-energy-based-retrieval-augmented-generat.md @@ -0,0 +1,67 @@ +# Paper: MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation + +--- +type: paper +title: "MODE-RAG: Manifold Outlier Diagnosis and Energy-based Retrieval-Augmented Generation Evaluation" +authors: Zehang Wei, Jiaxin Dai, Jiamin Yan, Xiang Xiang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17449 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.CV + - cs.LG + - cs.MM +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, rag, reasoning +- arXiv categories: cs.CL, cs.AI, cs.CV, cs.LG, cs.MM +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17449 diff --git a/papers/items/2026-2606-17453-mapsatisfybench-benchmarking-satisfaction-aware-map-agents-through-behavior-grou.md b/papers/items/2026-2606-17453-mapsatisfybench-benchmarking-satisfaction-aware-map-agents-through-behavior-grou.md new file mode 100644 index 0000000..58e4745 --- /dev/null +++ b/papers/items/2026-2606-17453-mapsatisfybench-benchmarking-satisfaction-aware-map-agents-through-behavior-grou.md @@ -0,0 +1,59 @@ +# Paper: MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors + +--- +type: paper +title: "MapSatisfyBench: Benchmarking Satisfaction-Aware Map Agents through Behavior-Grounded Implicit Decision Factors" +authors: Lubin Bai, Mengyu Cao, Sixue Wang, Zhongwei Wan, Yue Pan, Jiale Hou, Xiang Li, Xiuyuan Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17453 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17453 diff --git a/papers/items/2026-2606-17459-can-llms-be-ceos-benchmarking-strategic-resource-reallocation-with-multi-role-ag.md b/papers/items/2026-2606-17459-can-llms-be-ceos-benchmarking-strategic-resource-reallocation-with-multi-role-ag.md new file mode 100644 index 0000000..582477e --- /dev/null +++ b/papers/items/2026-2606-17459-can-llms-be-ceos-benchmarking-strategic-resource-reallocation-with-multi-role-ag.md @@ -0,0 +1,65 @@ +# Paper: Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation + +--- +type: paper +title: Can LLMs Be CEOs? Benchmarking Strategic Resource Reallocation with Multi-Role Agent Simulation +authors: Yuyang Dai, Xueqing Peng, Lingfei Qian, Zhuohan Xie +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17459 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - planning + - rag + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: agent-evaluation, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, planning-agent +- inferred topics: agent-evaluation, computer-use, multi-agent, planning, rag, reasoning, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17459 diff --git a/papers/items/2026-2606-17573-cordon-semantic-transactions-for-tool-using-llm-agents.md b/papers/items/2026-2606-17573-cordon-semantic-transactions-for-tool-using-llm-agents.md new file mode 100644 index 0000000..11f573d --- /dev/null +++ b/papers/items/2026-2606-17573-cordon-semantic-transactions-for-tool-using-llm-agents.md @@ -0,0 +1,63 @@ +# Paper: Cordon: Semantic Transactions for Tool-Using LLM Agents + +--- +type: paper +title: "Cordon: Semantic Transactions for Tool-Using LLM Agents" +authors: Zheng Chen, Hanqing Liu, Duling Xu, Dong Dong, Jialin Li, Bangzheng Pu, Jidong Zhai +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17573 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.OS + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, agent-safety, memory, tool-use, workflow-agent +- arXiv categories: cs.OS, cs.CR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17573 diff --git a/papers/items/2026-2606-17680-envrl-learn-from-environment-dynamics-in-agentic-reinforcement-learning.md b/papers/items/2026-2606-17680-envrl-learn-from-environment-dynamics-in-agentic-reinforcement-learning.md new file mode 100644 index 0000000..7c5d6cb --- /dev/null +++ b/papers/items/2026-2606-17680-envrl-learn-from-environment-dynamics-in-agentic-reinforcement-learning.md @@ -0,0 +1,62 @@ +# Paper: EnvRL: Learn from Environment Dynamics in Agentic Reinforcement Learning + +--- +type: paper +title: "EnvRL: Learn from Environment Dynamics in Agentic Reinforcement Learning" +authors: Zhitong Wang, Songze Li, Hao Peng, Shuzheng Si, Yi Wang, Maosong Sun, Juanzi Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.17680 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, planning, rag, tool-use +- arXiv categories: cs.LG, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.17680 diff --git a/papers/items/2026-2606-18023-loopcoder-v2-only-loop-once-for-efficient-test-time-computation-scaling.md b/papers/items/2026-2606-18023-loopcoder-v2-only-loop-once-for-efficient-test-time-computation-scaling.md new file mode 100644 index 0000000..b8fc4cb --- /dev/null +++ b/papers/items/2026-2606-18023-loopcoder-v2-only-loop-once-for-efficient-test-time-computation-scaling.md @@ -0,0 +1,63 @@ +# Paper: LoopCoder-v2: Only Loop Once for Efficient Test-Time Computation Scaling + +--- +type: paper +title: "LoopCoder-v2: Only Loop Once for Efficient Test-Time Computation Scaling" +authors: Jian Yang, Shawn Guo, Wei Zhang, Tianyu Zheng, Yaxin Du, Haau-Sing Li, Jiajun Wu, Yue Song, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18023 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, coding-agent, memory, reasoning, tool-use +- arXiv categories: cs.LG, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18023 diff --git a/papers/items/2026-2606-18037-provenanceguard-source-aware-factuality-verification-for-mcp-based-llm-agents.md b/papers/items/2026-2606-18037-provenanceguard-source-aware-factuality-verification-for-mcp-based-llm-agents.md new file mode 100644 index 0000000..c6a09c4 --- /dev/null +++ b/papers/items/2026-2606-18037-provenanceguard-source-aware-factuality-verification-for-mcp-based-llm-agents.md @@ -0,0 +1,64 @@ +# Paper: ProvenanceGuard: Source-Aware Factuality Verification for MCP-Based LLM Agents + +--- +type: paper +title: "ProvenanceGuard: Source-Aware Factuality Verification for MCP-Based LLM Agents" +authors: Ander Alvarez, Santhiya Rajan, Samuel Mugel, Román Orús +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18037 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, agent-safety, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18037 diff --git a/papers/items/2026-2606-18051-compositional-skill-routing-for-llm-agents-decompose-retrieve-and-compose.md b/papers/items/2026-2606-18051-compositional-skill-routing-for-llm-agents-decompose-retrieve-and-compose.md new file mode 100644 index 0000000..52b2ef5 --- /dev/null +++ b/papers/items/2026-2606-18051-compositional-skill-routing-for-llm-agents-decompose-retrieve-and-compose.md @@ -0,0 +1,61 @@ +# Paper: Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose + +--- +type: paper +title: "Compositional Skill Routing for LLM Agents: Decompose, Retrieve, and Compose" +authors: Xueping Gao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18051 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, tool-use +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18051 diff --git a/papers/items/2026-2606-18068-agentic-ai-based-framework-for-mitigating-premature-diagnostic-handoff-and-silen.md b/papers/items/2026-2606-18068-agentic-ai-based-framework-for-mitigating-premature-diagnostic-handoff-and-silen.md new file mode 100644 index 0000000..1dc7c01 --- /dev/null +++ b/papers/items/2026-2606-18068-agentic-ai-based-framework-for-mitigating-premature-diagnostic-handoff-and-silen.md @@ -0,0 +1,61 @@ +# Paper: Agentic AI-based Framework for Mitigating Premature Diagnostic Handoff and Silent Hallucination in Healthcare Applications + +--- +type: paper +title: Agentic AI-based Framework for Mitigating Premature Diagnostic Handoff and Silent Hallucination in Healthcare Applications +authors: Divyansh Srivastava, Shreya Ghosh, Anshul Verma, Rajkumar Buyya +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18068 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, agent-safety, multi-agent, reasoning +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18068 diff --git a/papers/items/2026-2606-18142-your-ai-travel-agent-would-book-you-a-bullfight-an-agentic-benchmark-for-implici.md b/papers/items/2026-2606-18142-your-ai-travel-agent-would-book-you-a-bullfight-an-agentic-benchmark-for-implici.md new file mode 100644 index 0000000..f15224d --- /dev/null +++ b/papers/items/2026-2606-18142-your-ai-travel-agent-would-book-you-a-bullfight-an-agentic-benchmark-for-implici.md @@ -0,0 +1,61 @@ +# Paper: Your AI Travel Agent Would Book You a Bullfight: An Agentic Benchmark for Implicit Animal Welfare in Frontier AI Models + +--- +type: paper +title: "Your AI Travel Agent Would Book You a Bullfight: An Agentic Benchmark for Implicit Animal Welfare in Frontier AI Models" +authors: Jasmine Brazilek, Joel Christoph, Maheep Chaudhary, Oliver Tullio, Carol Kline, Miles Tidmarsh, Arturs Kanepajs +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18142 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.CY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety +- arXiv categories: cs.AI, cs.CL, cs.CY +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18142 diff --git a/papers/items/2026-2606-18272-mitigating-anchoring-bias-in-llm-based-agents-for-energy-efficient-6g-autonomous.md b/papers/items/2026-2606-18272-mitigating-anchoring-bias-in-llm-based-agents-for-energy-efficient-6g-autonomous.md new file mode 100644 index 0000000..38badad --- /dev/null +++ b/papers/items/2026-2606-18272-mitigating-anchoring-bias-in-llm-based-agents-for-energy-efficient-6g-autonomous.md @@ -0,0 +1,63 @@ +# Paper: Mitigating Anchoring Bias in LLM-Based Agents for Energy-Efficient 6G Autonomous Networks + +--- +type: paper +title: Mitigating Anchoring Bias in LLM-Based Agents for Energy-Efficient 6G Autonomous Networks +authors: Hatim Chergui, Claudia Carballo González, Farhad Rezazadeh, Merouane Debbah +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18272 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-05 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.NI + - cs.AI + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-safety, computer-use, multi-agent, reasoning +- arXiv categories: cs.NI, cs.AI, eess.SY +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18272 diff --git a/papers/items/2026-2606-18356-safeclawbench-separating-semantic-audit-evidence-and-sandbox-harm-in-tool-using-.md b/papers/items/2026-2606-18356-safeclawbench-separating-semantic-audit-evidence-and-sandbox-harm-in-tool-using-.md new file mode 100644 index 0000000..a4ee243 --- /dev/null +++ b/papers/items/2026-2606-18356-safeclawbench-separating-semantic-audit-evidence-and-sandbox-harm-in-tool-using-.md @@ -0,0 +1,63 @@ +# Paper: SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents + +--- +type: paper +title: "SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents" +authors: Yuchuan Tian, Mengyu Zheng, Haocheng Mei, Ye Yuan, Chao Xu, Xinghao Chen, Hanting Chen, Yu Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18356 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-safety, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, tool-use +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18356 diff --git a/papers/items/2026-2606-18363-guava-an-effective-and-universal-harness-for-embodied-manipulation.md b/papers/items/2026-2606-18363-guava-an-effective-and-universal-harness-for-embodied-manipulation.md new file mode 100644 index 0000000..b1d19c9 --- /dev/null +++ b/papers/items/2026-2606-18363-guava-an-effective-and-universal-harness-for-embodied-manipulation.md @@ -0,0 +1,64 @@ +# Paper: Guava: An Effective and Universal Harness for Embodied Manipulation + +--- +type: paper +title: "Guava: An Effective and Universal Harness for Embodied Manipulation" +authors: Haowen Liu, Xirui Li, Shaoxiong Yao, Peng Shi, Tianyi Zhou, Jia-Bin Huang, Furong Huang, Jiayuan Mao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18363 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - embodied-agent + - planning + - reasoning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agentic-ai, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, tool-use +- inferred topics: embodied-agent, planning, reasoning, tool-use, workflow-agent, world-model +- arXiv categories: cs.RO, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18363 diff --git a/papers/items/2026-2606-18406-coremem-riemannian-retrieval-and-fisher-guided-distillation-for-long-term-memory.md b/papers/items/2026-2606-18406-coremem-riemannian-retrieval-and-fisher-guided-distillation-for-long-term-memory.md new file mode 100644 index 0000000..0f7bac4 --- /dev/null +++ b/papers/items/2026-2606-18406-coremem-riemannian-retrieval-and-fisher-guided-distillation-for-long-term-memory.md @@ -0,0 +1,63 @@ +# Paper: CoreMem: Riemannian Retrieval and Fisher-Guided Distillation for Long-Term Memory in Dialogue Agents + +--- +type: paper +title: "CoreMem: Riemannian Retrieval and Fisher-Guided Distillation for Long-Term Memory in Dialogue Agents" +authors: Jiaqi Chen, Yongqin Zeng, Shaoshen Chen, Yijian Zhang, Hai-Tao Zheng, Chunxia Ma, XiuTeng Zhou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18406 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18406 diff --git a/papers/items/2026-2606-18467-toolchain-crc-conformal-risk-control-for-agentic-ai-under-retrieval-and-tool-use.md b/papers/items/2026-2606-18467-toolchain-crc-conformal-risk-control-for-agentic-ai-under-retrieval-and-tool-use.md new file mode 100644 index 0000000..5a2da75 --- /dev/null +++ b/papers/items/2026-2606-18467-toolchain-crc-conformal-risk-control-for-agentic-ai-under-retrieval-and-tool-use.md @@ -0,0 +1,62 @@ +# Paper: ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift + +--- +type: paper +title: "ToolChain-CRC: Conformal Risk Control for Agentic AI Under Retrieval and Tool-Use Drift" +authors: Jeffery Opoku, David Banahene +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18467 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - stat.ML + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation, agentic-ai, ai-agent, rag-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, agentic-ai, ai-agent, rag-agent, tool-use +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: stat.ML, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18467 diff --git a/papers/items/2026-2606-18502-towards-scalable-customization-and-deployment-of-multi-agent-systems-for-enterpr.md b/papers/items/2026-2606-18502-towards-scalable-customization-and-deployment-of-multi-agent-systems-for-enterpr.md new file mode 100644 index 0000000..8feede3 --- /dev/null +++ b/papers/items/2026-2606-18502-towards-scalable-customization-and-deployment-of-multi-agent-systems-for-enterpr.md @@ -0,0 +1,62 @@ +# Paper: Towards Scalable Customization and Deployment of Multi-Agent Systems for Enterprise Applications + +--- +type: paper +title: Towards Scalable Customization and Deployment of Multi-Agent Systems for Enterprise Applications +authors: Paresh Dashore, Shreyas Kulkarni, Uttam Gurram, Nadia Bathaee, Kartik Balasubramaniam, Genta Indra Winata, Sambit Sahu, Shi-Xiong Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18502 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - coding-agent + - multi-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: coding-agent, multi-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18502 diff --git a/papers/items/2026-2606-18619-code-augur-agentic-vulnerability-detection-via-specification-inference.md b/papers/items/2026-2606-18619-code-augur-agentic-vulnerability-detection-via-specification-inference.md new file mode 100644 index 0000000..6f46cae --- /dev/null +++ b/papers/items/2026-2606-18619-code-augur-agentic-vulnerability-detection-via-specification-inference.md @@ -0,0 +1,63 @@ +# Paper: Code-Augur: Agentic Vulnerability Detection via Specification Inference + +--- +type: paper +title: "Code-Augur: Agentic Vulnerability Detection via Specification Inference" +authors: Zhengxiong Luo, Mehtab Zafar, Dylan Wolff, Abhik Roychoudhury +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18619 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-safety, computer-use, rag, reasoning +- arXiv categories: cs.CR, cs.AI, cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18619 diff --git a/papers/items/2026-2606-18671-hansel-extracting-breadcrumbs-from-web-agent-trajectories-for-interactive-verifi.md b/papers/items/2026-2606-18671-hansel-extracting-breadcrumbs-from-web-agent-trajectories-for-interactive-verifi.md new file mode 100644 index 0000000..73457d4 --- /dev/null +++ b/papers/items/2026-2606-18671-hansel-extracting-breadcrumbs-from-web-agent-trajectories-for-interactive-verifi.md @@ -0,0 +1,61 @@ +# Paper: HANSEL: Extracting Breadcrumbs from Web Agent Trajectories for Interactive Verification + +--- +type: paper +title: "HANSEL: Extracting Breadcrumbs from Web Agent Trajectories for Interactive Verification" +authors: Yujin Zhang, Daye Nam +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18671 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - planning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: ai-agent, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, web-gui-agent +- inferred topics: agent-evaluation, computer-use, embodied-agent, planning +- arXiv categories: cs.HC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18671 diff --git a/papers/items/2026-2606-18789-poweragentbench-ss-a-benchmark-for-agentic-ai-in-power-system-steady-state-studi.md b/papers/items/2026-2606-18789-poweragentbench-ss-a-benchmark-for-agentic-ai-in-power-system-steady-state-studi.md new file mode 100644 index 0000000..d6db767 --- /dev/null +++ b/papers/items/2026-2606-18789-poweragentbench-ss-a-benchmark-for-agentic-ai-in-power-system-steady-state-studi.md @@ -0,0 +1,64 @@ +# Paper: PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies + +--- +type: paper +title: "PowerAgentBench-SS: A Benchmark for Agentic AI in Power System Steady-State Studies" +authors: Costas Mylonas, Magda Foti, Andrea Pomarico, Matheus Duarte, Qian Zhang, Emmanouel Varvarigos +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18789 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 24 +collection_queries: agentic-ai, planning-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, planning-agent, tool-use +- inferred topics: agent-evaluation, agent-safety, computer-use, planning, rag, tool-use, workflow-agent +- arXiv categories: eess.SY +- collection score: 24 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18789 diff --git a/papers/items/2026-2606-18829-gatemem-benchmarking-memory-governance-in-multi-principal-shared-memory-agents.md b/papers/items/2026-2606-18829-gatemem-benchmarking-memory-governance-in-multi-principal-shared-memory-agents.md new file mode 100644 index 0000000..4d571b4 --- /dev/null +++ b/papers/items/2026-2606-18829-gatemem-benchmarking-memory-governance-in-multi-principal-shared-memory-agents.md @@ -0,0 +1,63 @@ +# Paper: GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents + +--- +type: paper +title: "GateMem: Benchmarking Memory Governance in Multi-Principal Shared-Memory Agents" +authors: Zhe Ren, Yibo Yang, Yimeng Chen, Zijun Zhao, Benshuo Fu, Zhihao Shu, Bingjie Zhang, Yangyang Xu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18829 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag, workflow-agent +- arXiv categories: cs.LG, cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18829 diff --git a/papers/items/2026-2606-18950-rtsgamebench-an-rts-benchmark-for-strategic-reasoning-by-vision-language-models.md b/papers/items/2026-2606-18950-rtsgamebench-an-rts-benchmark-for-strategic-reasoning-by-vision-language-models.md new file mode 100644 index 0000000..29c6558 --- /dev/null +++ b/papers/items/2026-2606-18950-rtsgamebench-an-rts-benchmark-for-strategic-reasoning-by-vision-language-models.md @@ -0,0 +1,64 @@ +# Paper: RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models + +--- +type: paper +title: "RTSGameBench: An RTS Benchmark for Strategic Reasoning by Vision-Language Models" +authors: San Kim, Daechul Ahn, Reokyoung Kim, Hyeonbeom Choi, Seungyeon Jwa, Jonghyun Choi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.18950 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, multi-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.18950 diff --git a/papers/items/2026-2606-19063-pypiline-malicious-pypi-package-detection-via-suspicious-api-knowledge-and-agent.md b/papers/items/2026-2606-19063-pypiline-malicious-pypi-package-detection-via-suspicious-api-knowledge-and-agent.md new file mode 100644 index 0000000..19945cd --- /dev/null +++ b/papers/items/2026-2606-19063-pypiline-malicious-pypi-package-detection-via-suspicious-api-knowledge-and-agent.md @@ -0,0 +1,62 @@ +# Paper: PYPILINE: Malicious PyPI Package Detection via Suspicious API Knowledge and Agent Workflow + +--- +type: paper +title: "PYPILINE: Malicious PyPI Package Detection via Suspicious API Knowledge and Agent Workflow" +authors: Siyuan Pang, Yepeng Yao, Zhengwei Jiang, Zijing Fan, Haozhe Li, Baoxu Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19063 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, rag-agent +- inferred topics: agent-evaluation, agent-safety, rag, tool-use, workflow-agent +- arXiv categories: cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19063 diff --git a/papers/items/2026-2606-19242-runtime-compliance-verification-for-ai-agents.md b/papers/items/2026-2606-19242-runtime-compliance-verification-for-ai-agents.md new file mode 100644 index 0000000..86760df --- /dev/null +++ b/papers/items/2026-2606-19242-runtime-compliance-verification-for-ai-agents.md @@ -0,0 +1,59 @@ +# Paper: Runtime Compliance Verification for AI Agents + +--- +type: paper +title: Runtime Compliance Verification for AI Agents +authors: Nafiseh Kahani, Masoud Barati, Diana Addae +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19242 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent, function-calling, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, function-calling, tool-use +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.SE +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19242 diff --git a/papers/items/2026-2606-19245-txbench-pp-analyzing-ai-agent-performance-on-small-molecule-preclinical-pharmaco.md b/papers/items/2026-2606-19245-txbench-pp-analyzing-ai-agent-performance-on-small-molecule-preclinical-pharmaco.md new file mode 100644 index 0000000..97a7473 --- /dev/null +++ b/papers/items/2026-2606-19245-txbench-pp-analyzing-ai-agent-performance-on-small-molecule-preclinical-pharmaco.md @@ -0,0 +1,64 @@ +# Paper: TxBench-PP: Analyzing AI Agent Performance on Small-Molecule Preclinical Pharmacology + +--- +type: paper +title: "TxBench-PP: Analyzing AI Agent Performance on Small-Molecule Preclinical Pharmacology" +authors: Hannah Le, Ramesh Ramasamy, Alex Urrutia, Mahsa Yazdani, Tim Proctor, Kenny Workman +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19245 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19245 diff --git a/papers/items/2026-2606-19409-openrath-session-centered-runtime-state-for-agent-systems.md b/papers/items/2026-2606-19409-openrath-session-centered-runtime-state-for-agent-systems.md new file mode 100644 index 0000000..7b7c480 --- /dev/null +++ b/papers/items/2026-2606-19409-openrath-session-centered-runtime-state-for-agent-systems.md @@ -0,0 +1,64 @@ +# Paper: OpenRath: Session-Centered Runtime State for Agent Systems + +--- +type: paper +title: "OpenRath: Session-Centered Runtime State for Agent Systems" +authors: Fukang Wen, Zhijie Wang, Ruilin Xu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19409 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.PL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, multi-agent, rag, tool-use, workflow-agent +- arXiv categories: cs.SE, cs.PL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19409 diff --git a/papers/items/2026-2606-19464-deontic-policies-for-runtime-governance-of-agentic-ai-systems.md b/papers/items/2026-2606-19464-deontic-policies-for-runtime-governance-of-agentic-ai-systems.md new file mode 100644 index 0000000..d367d35 --- /dev/null +++ b/papers/items/2026-2606-19464-deontic-policies-for-runtime-governance-of-agentic-ai-systems.md @@ -0,0 +1,63 @@ +# Paper: Deontic Policies for Runtime Governance of Agentic AI Systems + +--- +type: paper +title: Deontic Policies for Runtime Governance of Agentic AI Systems +authors: Anupam Joshi, Tim Finin, Karuna Pande Joshi, Lalana Kagal +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19464 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai, autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19464 diff --git a/papers/items/2026-2606-19613-staminabench-stress-testing-coding-agents-over-100-interaction-turns.md b/papers/items/2026-2606-19613-staminabench-stress-testing-coding-agents-over-100-interaction-turns.md new file mode 100644 index 0000000..541a3cf --- /dev/null +++ b/papers/items/2026-2606-19613-staminabench-stress-testing-coding-agents-over-100-interaction-turns.md @@ -0,0 +1,61 @@ +# Paper: StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns + +--- +type: paper +title: "StaminaBench: Stress-Testing Coding Agents over 100 Interaction Turns" +authors: Vlad Sobal, Shuo Yang, Yuting Zhang, Wei Xia, Stefano Soatto +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19613 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19613 diff --git a/papers/items/2026-2606-19704-beyond-static-leaderboards-predictive-validity-for-the-evaluation-of-llm-agents.md b/papers/items/2026-2606-19704-beyond-static-leaderboards-predictive-validity-for-the-evaluation-of-llm-agents.md new file mode 100644 index 0000000..def4fec --- /dev/null +++ b/papers/items/2026-2606-19704-beyond-static-leaderboards-predictive-validity-for-the-evaluation-of-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: Beyond Static Leaderboards: Predictive Validity for the Evaluation of LLM Agents + +--- +type: paper +title: "Beyond Static Leaderboards: Predictive Validity for the Evaluation of LLM Agents" +authors: Dhaval C. Patel, Kaoutar El Maghraoui, Shuxin Lin, Yusheng Li, Tianjun Feng, Chun-Yi Tsai, Yihan Sun, Wei Alexander Xin, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19704 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19704 diff --git a/papers/items/2026-2606-19787-oragentbench-can-llm-agents-solve-challenging-operations-research-tasks-end-to-e.md b/papers/items/2026-2606-19787-oragentbench-can-llm-agents-solve-challenging-operations-research-tasks-end-to-e.md new file mode 100644 index 0000000..eaf4e87 --- /dev/null +++ b/papers/items/2026-2606-19787-oragentbench-can-llm-agents-solve-challenging-operations-research-tasks-end-to-e.md @@ -0,0 +1,60 @@ +# Paper: ORAgentBench: Can LLM Agents Solve Challenging Operations Research Tasks End to End? + +--- +type: paper +title: "ORAgentBench: Can LLM Agents Solve Challenging Operations Research Tasks End to End?" +authors: Jiajun Li, Mingshu Cai, Yixuan Li, Yu Ding, Ran Hou, Guanyu Nie, Xiongwei Han, Wanyuan Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19787 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, rag, workflow-agent +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19787 diff --git a/papers/items/2026-2606-19812-human-on-the-loop-orchestration-for-ai-assisted-legal-discovery.md b/papers/items/2026-2606-19812-human-on-the-loop-orchestration-for-ai-assisted-legal-discovery.md new file mode 100644 index 0000000..c38f145 --- /dev/null +++ b/papers/items/2026-2606-19812-human-on-the-loop-orchestration-for-ai-assisted-legal-discovery.md @@ -0,0 +1,65 @@ +# Paper: Human-on-the-Loop Orchestration for AI-Assisted Legal Discovery + +--- +type: paper +title: Human-on-the-Loop Orchestration for AI-Assisted Legal Discovery +authors: Anushree Sinha, Srivaths Ranganathan, Abhishek Dharmaratnakar, Debanshu Das +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19812 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - reasoning + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, rag, reasoning, workflow-agent, world-model +- arXiv categories: cs.AI, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19812 diff --git a/papers/items/2026-2606-19852-prompt-plan-extract-zero-shot-agentic-llms-workflows-for-lung-pathology-extracti.md b/papers/items/2026-2606-19852-prompt-plan-extract-zero-shot-agentic-llms-workflows-for-lung-pathology-extracti.md new file mode 100644 index 0000000..5958c17 --- /dev/null +++ b/papers/items/2026-2606-19852-prompt-plan-extract-zero-shot-agentic-llms-workflows-for-lung-pathology-extracti.md @@ -0,0 +1,62 @@ +# Paper: Prompt, Plan, Extract: Zero-Shot Agentic LLMs Workflows for Lung Pathology Extraction from Clinical Narratives + +--- +type: paper +title: "Prompt, Plan, Extract: Zero-Shot Agentic LLMs Workflows for Lung Pathology Extraction from Clinical Narratives" +authors: Aman Pathak, Cheng Peng, Mengxian Lyu, Ziyi Chen, Reema Solan, Sankalp Talankar, Yasir Khan, Hiren Mehta, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19852 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, planning, tool-use, workflow-agent +- arXiv categories: cs.CL, cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19852 diff --git a/papers/items/2026-2606-19899-measuring-biological-capabilities-and-risks-of-ai-agents.md b/papers/items/2026-2606-19899-measuring-biological-capabilities-and-risks-of-ai-agents.md new file mode 100644 index 0000000..14aba93 --- /dev/null +++ b/papers/items/2026-2606-19899-measuring-biological-capabilities-and-risks-of-ai-agents.md @@ -0,0 +1,64 @@ +# Paper: Measuring Biological Capabilities and Risks of AI Agents + +--- +type: paper +title: Measuring Biological Capabilities and Risks of AI Agents +authors: Patricia Paskov, Jeffrey Lee, Kyle Brady, Alyssa Worland +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19899 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-evaluation, agentic-ai, ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, agentic-ai, ai-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, rag, tool-use, workflow-agent +- arXiv categories: cs.CY, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19899 diff --git a/papers/items/2026-2606-19926-memgui-agent-an-end-to-end-long-horizon-mobile-gui-agent-with-proactive-context-.md b/papers/items/2026-2606-19926-memgui-agent-an-end-to-end-long-horizon-mobile-gui-agent-with-proactive-context-.md new file mode 100644 index 0000000..cc3aad2 --- /dev/null +++ b/papers/items/2026-2606-19926-memgui-agent-an-end-to-end-long-horizon-mobile-gui-agent-with-proactive-context-.md @@ -0,0 +1,61 @@ +# Paper: MemGUI-Agent: An End-to-End Long-Horizon Mobile GUI Agent with Proactive Context Management + +--- +type: paper +title: "MemGUI-Agent: An End-to-End Long-Horizon Mobile GUI Agent with Proactive Context Management" +authors: Guangyi Liu, Gao Wu, Congxiao Liu, Pengxiang Zhao, Liang Liu, Mading Li, Qi Zhang, Mengyan Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19926 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use +- arXiv categories: cs.HC +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19926 diff --git a/papers/items/2026-2606-19930-mobileforge-annotation-free-adaptation-for-mobile-gui-agents-with-hierarchical-f.md b/papers/items/2026-2606-19930-mobileforge-annotation-free-adaptation-for-mobile-gui-agents-with-hierarchical-f.md new file mode 100644 index 0000000..ce1112d --- /dev/null +++ b/papers/items/2026-2606-19930-mobileforge-annotation-free-adaptation-for-mobile-gui-agents-with-hierarchical-f.md @@ -0,0 +1,60 @@ +# Paper: MobileForge: Annotation-Free Adaptation for Mobile GUI Agents with Hierarchical Feedback-Guided Policy Optimization + +--- +type: paper +title: "MobileForge: Annotation-Free Adaptation for Mobile GUI Agents with Hierarchical Feedback-Guided Policy Optimization" +authors: Guangyi Liu, Pengxiang Zhao, Gao Wu, Yiwen Yin, Mading Li, Liang Liu, Congxiao Liu, Zhang Qi, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19930 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, tool-use +- arXiv categories: cs.HC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19930 diff --git a/papers/items/2026-2606-19980-enpire-agentic-robot-policy-self-improvement-in-the-real-world.md b/papers/items/2026-2606-19980-enpire-agentic-robot-policy-self-improvement-in-the-real-world.md new file mode 100644 index 0000000..53472ae --- /dev/null +++ b/papers/items/2026-2606-19980-enpire-agentic-robot-policy-self-improvement-in-the-real-world.md @@ -0,0 +1,62 @@ +# Paper: ENPIRE: Agentic Robot Policy Self-Improvement in the Real World + +--- +type: paper +title: "ENPIRE: Agentic Robot Policy Self-Improvement in the Real World" +authors: "Wenli Xiao, Jia Xie, Tonghe Zhang, Haotian Lin, Letian \"Max\" Fu, Haoru Xue, Jalen Lu, Yi Yang, et al." +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.19980 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - embodied-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: coding-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, tool-use +- inferred topics: agent-evaluation, coding-agent, embodied-agent, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.19980 diff --git a/papers/items/2026-2606-20023-when-lower-privileges-suffice-investigating-over-privileged-tool-selection-in-ll.md b/papers/items/2026-2606-20023-when-lower-privileges-suffice-investigating-over-privileged-tool-selection-in-ll.md new file mode 100644 index 0000000..6c0f801 --- /dev/null +++ b/papers/items/2026-2606-20023-when-lower-privileges-suffice-investigating-over-privileged-tool-selection-in-ll.md @@ -0,0 +1,62 @@ +# Paper: When Lower Privileges Suffice: Investigating Over-Privileged Tool Selection in LLM Agents + +--- +type: paper +title: "When Lower Privileges Suffice: Investigating Over-Privileged Tool Selection in LLM Agents" +authors: Kaiyue Yang, Yuyan Bu, Jingwei Yi, Yuchi Wang, Biyu Zhou, Juntao Dai, Songlin Hu, Yaodong Yang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20023 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.SE, cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20023 diff --git a/papers/items/2026-2606-20041-ai-economist-agent-an-agentic-framework-for-model-grounded-economic-analysis-wit.md b/papers/items/2026-2606-20041-ai-economist-agent-an-agentic-framework-for-model-grounded-economic-analysis-wit.md new file mode 100644 index 0000000..5f99ce3 --- /dev/null +++ b/papers/items/2026-2606-20041-ai-economist-agent-an-agentic-framework-for-model-grounded-economic-analysis-wit.md @@ -0,0 +1,63 @@ +# Paper: AI Economist Agent: An Agentic Framework for Model-Grounded Economic Analysis with RAG, Knowledge Graphs, and Large Language Models + +--- +type: paper +title: "AI Economist Agent: An Agentic Framework for Model-Grounded Economic Analysis with RAG, Knowledge Graphs, and Large Language Models" +authors: Masahiro Kato +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20041 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - econ.GN + - cs.AI + - cs.LG + - q-fin.GN +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: ai-agent, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, rag-agent +- inferred topics: agent-evaluation, planning, rag +- arXiv categories: econ.GN, cs.AI, cs.LG, q-fin.GN +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20041 diff --git a/papers/items/2026-2606-20047-pacms-submodular-context-selection-as-a-pluggable-engine-for-llm-agents.md b/papers/items/2026-2606-20047-pacms-submodular-context-selection-as-a-pluggable-engine-for-llm-agents.md new file mode 100644 index 0000000..fa81355 --- /dev/null +++ b/papers/items/2026-2606-20047-pacms-submodular-context-selection-as-a-pluggable-engine-for-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: PACMS: Submodular Context Selection as a Pluggable Engine for LLM Agents + +--- +type: paper +title: "PACMS: Submodular Context Selection as a Pluggable Engine for LLM Agents" +authors: Manu Ghulyani, Arunabh Singh, Karan Bharadwaj, Ankit Nath, Suranjan Goswami +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20047 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.IR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20047 diff --git a/papers/items/2026-2606-20243-phoenix-safe-github-issue-resolution-via-multi-agent-llms.md b/papers/items/2026-2606-20243-phoenix-safe-github-issue-resolution-via-multi-agent-llms.md new file mode 100644 index 0000000..dea6699 --- /dev/null +++ b/papers/items/2026-2606-20243-phoenix-safe-github-issue-resolution-via-multi-agent-llms.md @@ -0,0 +1,64 @@ +# Paper: Phoenix: Safe GitHub Issue Resolution via Multi-Agent LLMs + +--- +type: paper +title: "Phoenix: Safe GitHub Issue Resolution via Multi-Agent LLMs" +authors: Kipngeno Koech, Muhammad Adam, Baimam Boukar Jean Jacques, Joao Barros +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20243 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - multi-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, multi-agent, planning, rag +- arXiv categories: cs.SE, cs.MA +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20243 diff --git a/papers/items/2026-2606-20401-poweragentbench-dyn-a-benchmark-for-agentic-ai-in-power-system-dynamic-studies.md b/papers/items/2026-2606-20401-poweragentbench-dyn-a-benchmark-for-agentic-ai-in-power-system-dynamic-studies.md new file mode 100644 index 0000000..c3d6168 --- /dev/null +++ b/papers/items/2026-2606-20401-poweragentbench-dyn-a-benchmark-for-agentic-ai-in-power-system-dynamic-studies.md @@ -0,0 +1,66 @@ +# Paper: PowerAgentBench-Dyn: A Benchmark for Agentic AI in Power System Dynamic Studies + +--- +type: paper +title: "PowerAgentBench-Dyn: A Benchmark for Agentic AI in Power System Dynamic Studies" +authors: Qian Zhang, Andrea Pomarico, Costas Mylonas, Magda Foti, Alberto Berizzi, Le Xie +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20401 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - planning + - rag + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 24 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, planning, rag, reasoning, tool-use, world-model +- arXiv categories: eess.SY +- collection score: 24 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20401 diff --git a/papers/items/2026-2606-20470-analyzing-defensive-misdirection-against-model-guided-automated-attacks-on-agent.md b/papers/items/2026-2606-20470-analyzing-defensive-misdirection-against-model-guided-automated-attacks-on-agent.md new file mode 100644 index 0000000..a223886 --- /dev/null +++ b/papers/items/2026-2606-20470-analyzing-defensive-misdirection-against-model-guided-automated-attacks-on-agent.md @@ -0,0 +1,62 @@ +# Paper: Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems + +--- +type: paper +title: Analyzing Defensive Misdirection Against Model-Guided Automated Attacks on Agentic AI Systems +authors: Reza Soosahabi, Vivek Namsani +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20470 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, computer-use, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20470 diff --git a/papers/items/2026-2606-20479-groundcontrol-anticipating-navigation-failures-in-vision-language-agents-via-tra.md b/papers/items/2026-2606-20479-groundcontrol-anticipating-navigation-failures-in-vision-language-agents-via-tra.md new file mode 100644 index 0000000..f34626a --- /dev/null +++ b/papers/items/2026-2606-20479-groundcontrol-anticipating-navigation-failures-in-vision-language-agents-via-tra.md @@ -0,0 +1,62 @@ +# Paper: GroundControl: Anticipating Navigation Failures in Vision-Language Agents via Trajectory-Consistent Uncertainty Estimates + +--- +type: paper +title: "GroundControl: Anticipating Navigation Failures in Vision-Language Agents via Trajectory-Consistent Uncertainty Estimates" +authors: Nastaran Darabi, Divake Kumar, Sina Tayebati, Devashri Naik, Amit Ranjan Trivedi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20479 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - embodied-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, agent-safety, embodied-agent, rag, tool-use +- arXiv categories: cs.RO +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20479 diff --git a/papers/items/2026-2606-20510-efficient-and-sound-probabilistic-verification-for-ai-agents.md b/papers/items/2026-2606-20510-efficient-and-sound-probabilistic-verification-for-ai-agents.md new file mode 100644 index 0000000..426d7bf --- /dev/null +++ b/papers/items/2026-2606-20510-efficient-and-sound-probabilistic-verification-for-ai-agents.md @@ -0,0 +1,62 @@ +# Paper: Efficient and Sound Probabilistic Verification for AI Agents + +--- +type: paper +title: Efficient and Sound Probabilistic Verification for AI Agents +authors: Alaia Solko-Breslin, Pramod Kaushik Mudrakarta, Mihai Christodorescu, Somesh Jha, Krishnamurthy Dj Dvijotham +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20510 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20510 diff --git a/papers/items/2026-2606-20512-probe-and-refine-tuning-of-repository-guidance-for-coding-agents.md b/papers/items/2026-2606-20512-probe-and-refine-tuning-of-repository-guidance-for-coding-agents.md new file mode 100644 index 0000000..24a1b8b --- /dev/null +++ b/papers/items/2026-2606-20512-probe-and-refine-tuning-of-repository-guidance-for-coding-agents.md @@ -0,0 +1,64 @@ +# Paper: Probe-and-Refine Tuning of Repository Guidance for Coding Agents + +--- +type: paper +title: Probe-and-Refine Tuning of Repository Guidance for Coding Agents +authors: Asa Shepard, Jeannie Albrecht +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20512 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: coding-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, tool-use +- inferred topics: agent-evaluation, coding-agent, computer-use, rag, tool-use, workflow-agent +- arXiv categories: cs.SE, cs.LG +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20512 diff --git a/papers/items/2026-2606-20515-s-agent-spatial-tool-use-elicits-reasoning-for-spatial-intelligence.md b/papers/items/2026-2606-20515-s-agent-spatial-tool-use-elicits-reasoning-for-spatial-intelligence.md new file mode 100644 index 0000000..4ea5553 --- /dev/null +++ b/papers/items/2026-2606-20515-s-agent-spatial-tool-use-elicits-reasoning-for-spatial-intelligence.md @@ -0,0 +1,62 @@ +# Paper: S-Agent: Spatial Tool-Use Elicits Reasoning for Spatial Intelligence + +--- +type: paper +title: "S-Agent: Spatial Tool-Use Elicits Reasoning for Spatial Intelligence" +authors: Yalun Dai, Hao Li, Shulin Tian, Runmao Yao, Yuhao Dong, Fangzhou Hong, Zhaoxi Chen, Fangfu Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20515 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, tool-use +- inferred topics: agent-evaluation, memory, planning, reasoning, tool-use +- arXiv categories: cs.CV +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20515 diff --git a/papers/items/2026-2606-20573-aona-a-comprehensive-architecture-and-workflow-design-for-global-agentic-collabo.md b/papers/items/2026-2606-20573-aona-a-comprehensive-architecture-and-workflow-design-for-global-agentic-collabo.md new file mode 100644 index 0000000..dc4a64d --- /dev/null +++ b/papers/items/2026-2606-20573-aona-a-comprehensive-architecture-and-workflow-design-for-global-agentic-collabo.md @@ -0,0 +1,61 @@ +# Paper: AONA: A Comprehensive Architecture and Workflow Design for Global Agentic Collaboration + +--- +type: paper +title: "AONA: A Comprehensive Architecture and Workflow Design for Global Agentic Collaboration" +authors: Jinliang Xu, Runkai Zhu, Bingqi Li, Fanjie Nie, Jin Li, Jiagui Xie +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20573 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-04-30 +updated_at: 2026-04-30 +status: queued +relevance: high +topics: + - multi-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.NI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: multi-agent, tool-use, workflow-agent +- arXiv categories: cs.NI, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20573 diff --git a/papers/items/2026-2606-20629-specialize-roles-mix-deployments-pushing-the-cost-accuracy-frontier-of-llm-agent.md b/papers/items/2026-2606-20629-specialize-roles-mix-deployments-pushing-the-cost-accuracy-frontier-of-llm-agent.md new file mode 100644 index 0000000..f8e54e6 --- /dev/null +++ b/papers/items/2026-2606-20629-specialize-roles-mix-deployments-pushing-the-cost-accuracy-frontier-of-llm-agent.md @@ -0,0 +1,65 @@ +# Paper: Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams + +--- +type: paper +title: "Specialize Roles, Mix Deployments: Pushing the Cost-Accuracy Frontier of LLM Agent Teams" +authors: Yinsicheng Jiang, Liang Cheng, Yeqi Huang, Yufan Zhao, Zhan Lu, Li Dong, Wenda Li, Edoardo Ponti, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20629 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-05-28 +updated_at: 2026-05-28 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, computer-use, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.MA, cs.AI, cs.LG +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20629 diff --git a/papers/items/2026-2606-20717-mirage-stealthy-visual-prompt-injection-for-vulnerability-detection-in-web-agent.md b/papers/items/2026-2606-20717-mirage-stealthy-visual-prompt-injection-for-vulnerability-detection-in-web-agent.md new file mode 100644 index 0000000..509440d --- /dev/null +++ b/papers/items/2026-2606-20717-mirage-stealthy-visual-prompt-injection-for-vulnerability-detection-in-web-agent.md @@ -0,0 +1,65 @@ +# Paper: MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents + +--- +type: paper +title: "MIRAGE: Stealthy Visual Prompt Injection for Vulnerability Detection in Web Agents" +authors: Xuelong Dai, Jianyu Ma, Boyang Ma, Biwei Yan, Yijun Yang, Yue Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20717 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-16 +updated_at: 2026-06-16 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, rag, tool-use, workflow-agent +- arXiv categories: cs.CV, cs.AI, cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20717 diff --git a/papers/items/2026-2606-20785-fara-1-5-scalable-learning-environments-for-computer-use-agents.md b/papers/items/2026-2606-20785-fara-1-5-scalable-learning-environments-for-computer-use-agents.md new file mode 100644 index 0000000..62244dd --- /dev/null +++ b/papers/items/2026-2606-20785-fara-1-5-scalable-learning-environments-for-computer-use-agents.md @@ -0,0 +1,63 @@ +# Paper: Fara-1.5: Scalable Learning Environments for Computer Use Agents + +--- +type: paper +title: "Fara-1.5: Scalable Learning Environments for Computer Use Agents" +authors: Ahmed Awadallah, Sahil Gupta, Yash Lara, Yadong Lu, Hussein Mozannar, Akshay Nambi, Zach Nussbaum, Yash Pandya, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20785 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20785 diff --git a/papers/items/2026-2606-20922-think-twice-before-you-act-protecting-llm-agents-against-tool-description-poison.md b/papers/items/2026-2606-20922-think-twice-before-you-act-protecting-llm-agents-against-tool-description-poison.md new file mode 100644 index 0000000..c73b637 --- /dev/null +++ b/papers/items/2026-2606-20922-think-twice-before-you-act-protecting-llm-agents-against-tool-description-poison.md @@ -0,0 +1,61 @@ +# Paper: Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning + +--- +type: paper +title: "Think Twice Before You Act: Protecting LLM Agents Against Tool Description Poisoning via Isolated Planning" +authors: Shanghao Shi, Xiao Wang, Chaoyu Zhang, Hao Li, Wenjing Lou, Thomas Hou, Yevgeniy Vorobeychik, Chongjie Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20922 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, tool-use +- arXiv categories: cs.CR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20922 diff --git a/papers/items/2026-2606-20950-power-systems-agent-benchmark-executable-evaluation-of-ai-agents-in-electric-pow.md b/papers/items/2026-2606-20950-power-systems-agent-benchmark-executable-evaluation-of-ai-agents-in-electric-pow.md new file mode 100644 index 0000000..7d3ddba --- /dev/null +++ b/papers/items/2026-2606-20950-power-systems-agent-benchmark-executable-evaluation-of-ai-agents-in-electric-pow.md @@ -0,0 +1,61 @@ +# Paper: Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering + +--- +type: paper +title: "Power Systems Agent Benchmark: Executable Evaluation of AI Agents in Electric Power Engineering" +authors: Sergei Trashchenkov +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20950 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation, ai-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, ai-agent, tool-use +- inferred topics: agent-evaluation, rag, tool-use +- arXiv categories: cs.AI, eess.SY +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20950 diff --git a/papers/items/2026-2606-20954-learning-what-not-to-forget-long-horizon-agent-memory-from-a-few-kilobytes-of-le.md b/papers/items/2026-2606-20954-learning-what-not-to-forget-long-horizon-agent-memory-from-a-few-kilobytes-of-le.md new file mode 100644 index 0000000..8bb167f --- /dev/null +++ b/papers/items/2026-2606-20954-learning-what-not-to-forget-long-horizon-agent-memory-from-a-few-kilobytes-of-le.md @@ -0,0 +1,62 @@ +# Paper: Learning What Not to Forget: Long-Horizon Agent Memory from a Few Kilobytes of Learning + +--- +type: paper +title: "Learning What Not to Forget: Long-Horizon Agent Memory from a Few Kilobytes of Learning" +authors: Nusrat Jahan Lia, Aritra Mazumder +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.20954 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-18 +updated_at: 2026-06-18 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.20954 diff --git a/papers/items/2026-2606-21013-agentic-time-machine-as-an-infrastructure-for-future-event-forecasting.md b/papers/items/2026-2606-21013-agentic-time-machine-as-an-infrastructure-for-future-event-forecasting.md new file mode 100644 index 0000000..c856d51 --- /dev/null +++ b/papers/items/2026-2606-21013-agentic-time-machine-as-an-infrastructure-for-future-event-forecasting.md @@ -0,0 +1,63 @@ +# Paper: Agentic Time Machine as an Infrastructure for Future-Event Forecasting + +--- +type: paper +title: Agentic Time Machine as an Infrastructure for Future-Event Forecasting +authors: Jingyi Chai, Bingyang Zheng, Xiangrui Liu, Hao Lu, Zihang Zhou, Tianchen Wang, Kemeng Zhang, Siheng Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21013 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, multi-agent, planning, rag, tool-use +- arXiv categories: cs.AI, cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21013 diff --git a/papers/items/2026-2606-21123-a-multi-agent-audit-framework-for-high-stakes-reasoning-evaluation-and-interpret.md b/papers/items/2026-2606-21123-a-multi-agent-audit-framework-for-high-stakes-reasoning-evaluation-and-interpret.md new file mode 100644 index 0000000..7241177 --- /dev/null +++ b/papers/items/2026-2606-21123-a-multi-agent-audit-framework-for-high-stakes-reasoning-evaluation-and-interpret.md @@ -0,0 +1,62 @@ +# Paper: A Multi-Agent Audit Framework for High-Stakes Reasoning: Evaluation and Interpretability in Clinical Mental Health Screening + +--- +type: paper +title: "A Multi-Agent Audit Framework for High-Stakes Reasoning: Evaluation and Interpretability in Clinical Mental Health Screening" +authors: Jingchen Ye, Yanpei Yu, Luyao Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21123 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, multi-agent, rag, reasoning, workflow-agent +- arXiv categories: cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21123 diff --git a/papers/items/2026-2606-21129-agenticos-an-intent-oriented-secure-operating-system-architecture-for-autonomous.md b/papers/items/2026-2606-21129-agenticos-an-intent-oriented-secure-operating-system-architecture-for-autonomous.md new file mode 100644 index 0000000..96a7d13 --- /dev/null +++ b/papers/items/2026-2606-21129-agenticos-an-intent-oriented-secure-operating-system-architecture-for-autonomous.md @@ -0,0 +1,61 @@ +# Paper: AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents + +--- +type: paper +title: "AgenticOS: An Intent-Oriented Secure Operating System Architecture for Autonomous AI Agents" +authors: Zhen Zhao, Yu Zhang, Yanpeng Zhu, Jia Wang, Songqiao Tao, Xin Cheng, Jiexin Gao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21129 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-safety + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.OS +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: ai-agent, autonomous-agent-llm, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, autonomous-agent-llm, tool-use +- inferred topics: agent-safety, planning, tool-use +- arXiv categories: cs.CR, cs.OS +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21129 diff --git a/papers/items/2026-2606-21228-sakana-fugu-technical-report.md b/papers/items/2026-2606-21228-sakana-fugu-technical-report.md new file mode 100644 index 0000000..31acba7 --- /dev/null +++ b/papers/items/2026-2606-21228-sakana-fugu-technical-report.md @@ -0,0 +1,61 @@ +# Paper: Sakana Fugu Technical Report + +--- +type: paper +title: Sakana Fugu Technical Report +authors: Yujin Tang, Edoardo Cetin, Jinglue Xu, Qi Sun, Stefan Nielsen, Vincent Richard, Haruto Goda, Iaroslav Tymchenko, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21228 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - coding-agent + - multi-agent + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: coding-agent, multi-agent, rag, reasoning +- arXiv categories: cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21228 diff --git a/papers/items/2026-2606-21401-swarmx-agentic-scheduling-for-low-latency-agentic-systems.md b/papers/items/2026-2606-21401-swarmx-agentic-scheduling-for-low-latency-agentic-systems.md new file mode 100644 index 0000000..03de065 --- /dev/null +++ b/papers/items/2026-2606-21401-swarmx-agentic-scheduling-for-low-latency-agentic-systems.md @@ -0,0 +1,62 @@ +# Paper: SwarmX: Agentic Scheduling for Low-Latency Agentic Systems + +--- +type: paper +title: "SwarmX: Agentic Scheduling for Low-Latency Agentic Systems" +authors: Yeqi Huang, Yanwei Ye, Guomin Chen, Wenhao Su, Bin Gong, Jialian Li, Zhan Lu, Yangshen Deng, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21401 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DC + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, multi-agent, tool-use, workflow-agent +- arXiv categories: cs.DC, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21401 diff --git a/papers/items/2026-2606-21409-don-t-blindly-trust-it-how-unreliable-feedback-breaks-tool-using-llm-agents.md b/papers/items/2026-2606-21409-don-t-blindly-trust-it-how-unreliable-feedback-breaks-tool-using-llm-agents.md new file mode 100644 index 0000000..cd843a4 --- /dev/null +++ b/papers/items/2026-2606-21409-don-t-blindly-trust-it-how-unreliable-feedback-breaks-tool-using-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: Don't Blindly Trust It: How Unreliable Feedback Breaks Tool-Using LLM Agents + +--- +type: paper +title: "Don't Blindly Trust It: How Unreliable Feedback Breaks Tool-Using LLM Agents" +authors: Chubin Zhang, Zhenglin Wan, Xingrui Yu, Pengfei Zhou, Wangbo Zhao, Jingxuan Wu, Yaxin Zhou, Ivor Tsang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21409 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, coding-agent, rag, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21409 diff --git a/papers/items/2026-2606-21445-autoras-learning-robust-agentic-systems-with-primitive-representations.md b/papers/items/2026-2606-21445-autoras-learning-robust-agentic-systems-with-primitive-representations.md new file mode 100644 index 0000000..c7ef1cf --- /dev/null +++ b/papers/items/2026-2606-21445-autoras-learning-robust-agentic-systems-with-primitive-representations.md @@ -0,0 +1,62 @@ +# Paper: AutoRAS: Learning Robust Agentic Systems with Primitive Representations + +--- +type: paper +title: "AutoRAS: Learning Robust Agentic Systems with Primitive Representations" +authors: Yang Yue, Xuancheng Zhu, Yuyang Ma, Guoshun Nan, Zihan Dou, Jingru Shan, Congyu Guo, Ji Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21445 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-safety + - multi-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-safety, multi-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21445 diff --git a/papers/items/2026-2606-21553-dissecting-agentic-rag-a-component-ablation-for-multi-hop-qa-with-a-local-7b-mod.md b/papers/items/2026-2606-21553-dissecting-agentic-rag-a-component-ablation-for-multi-hop-qa-with-a-local-7b-mod.md new file mode 100644 index 0000000..96ff47f --- /dev/null +++ b/papers/items/2026-2606-21553-dissecting-agentic-rag-a-component-ablation-for-multi-hop-qa-with-a-local-7b-mod.md @@ -0,0 +1,62 @@ +# Paper: Dissecting Agentic RAG: A Component Ablation for Multi-Hop QA with a Local 7B Model + +--- +type: paper +title: "Dissecting Agentic RAG: A Component Ablation for Multi-Hop QA with a Local 7B Model" +authors: Sheroz Shaikh +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21553 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, rag, reasoning, tool-use +- arXiv categories: cs.CL, cs.IR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21553 diff --git a/papers/items/2026-2606-21565-composing-verifiable-conceptual-models-via-building-blocks-towards-design-time-v.md b/papers/items/2026-2606-21565-composing-verifiable-conceptual-models-via-building-blocks-towards-design-time-v.md new file mode 100644 index 0000000..d605684 --- /dev/null +++ b/papers/items/2026-2606-21565-composing-verifiable-conceptual-models-via-building-blocks-towards-design-time-v.md @@ -0,0 +1,62 @@ +# Paper: Composing Verifiable Conceptual Models via Building Blocks: Towards Design-Time Verification of Agentic AI Workflows + +--- +type: paper +title: "Composing Verifiable Conceptual Models via Building Blocks: Towards Design-Time Verification of Agentic AI Workflows" +authors: Noe Y. Flandre, Alexander C. Nwala, Philippe J. Giabbanelli +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21565 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, reasoning, tool-use, workflow-agent, world-model +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21565 diff --git a/papers/items/2026-2606-21627-counsel-a-meta-evaluation-dataset-for-agentic-tasks.md b/papers/items/2026-2606-21627-counsel-a-meta-evaluation-dataset-for-agentic-tasks.md new file mode 100644 index 0000000..4f27e82 --- /dev/null +++ b/papers/items/2026-2606-21627-counsel-a-meta-evaluation-dataset-for-agentic-tasks.md @@ -0,0 +1,62 @@ +# Paper: Counsel: A Meta-Evaluation Dataset for Agentic Tasks + +--- +type: paper +title: "Counsel: A Meta-Evaluation Dataset for Agentic Tasks" +authors: Sashank Pisupati, Henry Broomfield, Eujeong Choi, Antonia Calvi, Charlie Wang, Roman Engeler, Max Bartolo, Patrick Lewis +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21627 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-evaluation, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, reasoning +- arXiv categories: cs.AI, cs.LG +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21627 diff --git a/papers/items/2026-2606-21649-evoembedding-evolvable-representations-for-long-context-retrieval-and-agentic-me.md b/papers/items/2026-2606-21649-evoembedding-evolvable-representations-for-long-context-retrieval-and-agentic-me.md new file mode 100644 index 0000000..a30dede --- /dev/null +++ b/papers/items/2026-2606-21649-evoembedding-evolvable-representations-for-long-context-retrieval-and-agentic-me.md @@ -0,0 +1,62 @@ +# Paper: EvoEmbedding: Evolvable Representations for Long-Context Retrieval and Agentic Memory + +--- +type: paper +title: "EvoEmbedding: Evolvable Representations for Long-Context Retrieval and Agentic Memory" +authors: Chang Nie, Chaoyou Fu, Junlan Feng, Caifeng Shan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21649 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory, agentic-ai, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, agentic-ai, rag-agent +- inferred topics: agent-evaluation, coding-agent, memory, rag, workflow-agent +- arXiv categories: cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21649 diff --git a/papers/items/2026-2606-21710-privacyalign-contextual-privacy-alignment-for-llm-agents.md b/papers/items/2026-2606-21710-privacyalign-contextual-privacy-alignment-for-llm-agents.md new file mode 100644 index 0000000..bd7f06e --- /dev/null +++ b/papers/items/2026-2606-21710-privacyalign-contextual-privacy-alignment-for-llm-agents.md @@ -0,0 +1,63 @@ +# Paper: PrivacyAlign: Contextual Privacy Alignment for LLM Agents + +--- +type: paper +title: "PrivacyAlign: Contextual Privacy Alignment for LLM Agents" +authors: Manveer Singh Tamber, Abhay Puri, Marc-Etienne Brunet, Perouz Taslakian, Jimmy Lin, Spandana Gella +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21710 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, planning, tool-use +- arXiv categories: cs.CL, cs.AI, cs.IR +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21710 diff --git a/papers/items/2026-2606-21732-safe-to-check-unsafe-to-use-relinking-at-the-compression-boundary-of-llm-agents.md b/papers/items/2026-2606-21732-safe-to-check-unsafe-to-use-relinking-at-the-compression-boundary-of-llm-agents.md new file mode 100644 index 0000000..61dd2f8 --- /dev/null +++ b/papers/items/2026-2606-21732-safe-to-check-unsafe-to-use-relinking-at-the-compression-boundary-of-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents + +--- +type: paper +title: "Safe to Check, Unsafe to Use: Relinking at the Compression Boundary of LLM Agents" +authors: Zesen Liu, Zihan Zhang, Dongdong She +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21732 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21732 diff --git a/papers/items/2026-2606-21740-training-the-orchestrator-a-supervised-approach-to-end-to-end-pddl-planning-with.md b/papers/items/2026-2606-21740-training-the-orchestrator-a-supervised-approach-to-end-to-end-pddl-planning-with.md new file mode 100644 index 0000000..1bb4457 --- /dev/null +++ b/papers/items/2026-2606-21740-training-the-orchestrator-a-supervised-approach-to-end-to-end-pddl-planning-with.md @@ -0,0 +1,62 @@ +# Paper: Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents + +--- +type: paper +title: "Training the Orchestrator: A Supervised Approach to End-to-End PDDL Planning with LLM Agents" +authors: Rajesh Mangannavar, Zachary Coalson, Pranay Dugar, Prasad Tadepalli +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21740 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-19 +updated_at: 2026-06-19 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, computer-use, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21740 diff --git a/papers/items/2026-2606-21836-agentdse-reasoning-augmented-architectural-design-space-exploration.md b/papers/items/2026-2606-21836-agentdse-reasoning-augmented-architectural-design-space-exploration.md new file mode 100644 index 0000000..ee7e384 --- /dev/null +++ b/papers/items/2026-2606-21836-agentdse-reasoning-augmented-architectural-design-space-exploration.md @@ -0,0 +1,62 @@ +# Paper: AgentDSE: Reasoning-Augmented Architectural Design Space Exploration + +--- +type: paper +title: "AgentDSE: Reasoning-Augmented Architectural Design Space Exploration" +authors: Chenyu Wang, Jiahe Caroline Shi, David Kong, Duane S. Boning, Zishen Wan, Yilun Du, Vijay Janapa Reddi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21836 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-20 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, reasoning +- arXiv categories: cs.AR, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21836 diff --git a/papers/items/2026-2606-21842-agent-assisted-side-channel-attacks-on-non-prefix-kv-cache-in-rag.md b/papers/items/2026-2606-21842-agent-assisted-side-channel-attacks-on-non-prefix-kv-cache-in-rag.md new file mode 100644 index 0000000..15260d0 --- /dev/null +++ b/papers/items/2026-2606-21842-agent-assisted-side-channel-attacks-on-non-prefix-kv-cache-in-rag.md @@ -0,0 +1,62 @@ +# Paper: Agent-Assisted Side-Channel Attacks on Non-Prefix KV Cache in RAG + +--- +type: paper +title: Agent-Assisted Side-Channel Attacks on Non-Prefix KV Cache in RAG +authors: He Sun, Shinan Liu, Siyuan Ma, Junhao Li, Mingjun Xiao, Wenhao Jiang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21842 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-20 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use +- arXiv categories: cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21842 diff --git a/papers/items/2026-2606-21877-agentriskbom-a-risk-scoping-security-bill-of-materials-for-agentic-ai-systems.md b/papers/items/2026-2606-21877-agentriskbom-a-risk-scoping-security-bill-of-materials-for-agentic-ai-systems.md new file mode 100644 index 0000000..c57740b --- /dev/null +++ b/papers/items/2026-2606-21877-agentriskbom-a-risk-scoping-security-bill-of-materials-for-agentic-ai-systems.md @@ -0,0 +1,66 @@ +# Paper: AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems + +--- +type: paper +title: "AgentRiskBOM: A Risk-Scoping Security Bill of Materials for Agentic AI Systems" +authors: Srimonti Dutta, Akshata Kishore Moharir +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21877 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-20 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CR + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: agentic-ai, ai-agent, rag-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, ai-agent, rag-agent, tool-use +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, multi-agent, rag, tool-use +- arXiv categories: cs.AI, cs.CR, cs.SE +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21877 diff --git a/papers/items/2026-2606-21963-holmes-multimodal-agentic-diagnosis-for-mixed-language-mobile-crashes-at-industr.md b/papers/items/2026-2606-21963-holmes-multimodal-agentic-diagnosis-for-mixed-language-mobile-crashes-at-industr.md new file mode 100644 index 0000000..13b812e --- /dev/null +++ b/papers/items/2026-2606-21963-holmes-multimodal-agentic-diagnosis-for-mixed-language-mobile-crashes-at-industr.md @@ -0,0 +1,63 @@ +# Paper: Holmes: Multimodal Agentic Diagnosis for Mixed-Language Mobile Crashes at Industrial Scale + +--- +type: paper +title: "Holmes: Multimodal Agentic Diagnosis for Mixed-Language Mobile Crashes at Industrial Scale" +authors: Jia Li, Wenyuan Ma, Ting Peng, Haibin Zheng, Yuetang Deng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.21963 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-20 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, computer-use, multi-agent, rag, workflow-agent +- arXiv categories: cs.AI, cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.21963 diff --git a/papers/items/2026-2606-22030-nous-a-predictive-world-model-for-long-term-agent-memory.md b/papers/items/2026-2606-22030-nous-a-predictive-world-model-for-long-term-agent-memory.md new file mode 100644 index 0000000..b740456 --- /dev/null +++ b/papers/items/2026-2606-22030-nous-a-predictive-world-model-for-long-term-agent-memory.md @@ -0,0 +1,64 @@ +# Paper: Nous: A Predictive World Model for Long-Term Agent Memory + +--- +type: paper +title: "Nous: A Predictive World Model for Long-Term Agent Memory" +authors: Pranav Singh +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22030 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-20 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.IR + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, world-model +- arXiv categories: cs.AI, cs.CL, cs.IR, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22030 diff --git a/papers/items/2026-2606-22082-codeteam-an-llm-powered-multi-agent-framework-for-repository-level-code-generati.md b/papers/items/2026-2606-22082-codeteam-an-llm-powered-multi-agent-framework-for-repository-level-code-generati.md new file mode 100644 index 0000000..9015b3d --- /dev/null +++ b/papers/items/2026-2606-22082-codeteam-an-llm-powered-multi-agent-framework-for-repository-level-code-generati.md @@ -0,0 +1,63 @@ +# Paper: CodeTeam: An LLM-Powered Multi-Agent Framework for Repository-Level Code Generation + +--- +type: paper +title: "CodeTeam: An LLM-Powered Multi-Agent Framework for Repository-Level Code Generation" +authors: Yifei Wang, Ruiyin Li, Peng Liang, Qiong Feng, Zengyang Li, Mojtaba Shahin, Arif Ali Khan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22082 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-20 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent, planning, rag +- arXiv categories: cs.SE, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22082 diff --git a/papers/items/2026-2606-22110-traceview-interactive-visualization-of-agentic-program-repair-trajectories.md b/papers/items/2026-2606-22110-traceview-interactive-visualization-of-agentic-program-repair-trajectories.md new file mode 100644 index 0000000..d86416b --- /dev/null +++ b/papers/items/2026-2606-22110-traceview-interactive-visualization-of-agentic-program-repair-trajectories.md @@ -0,0 +1,64 @@ +# Paper: TraceView: Interactive Visualization of Agentic Program Repair Trajectories + +--- +type: paper +title: "TraceView: Interactive Visualization of Agentic Program Repair Trajectories" +authors: Amirali Sajadi, Tu Nguyen, Kimmie Huynh, Esteban Parra, Preetha Chatterjee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22110 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-20 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, coding-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.SE, cs.AI, cs.HC +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22110 diff --git a/papers/items/2026-2606-22151-novelty-aware-agentic-retrieval-comparing-research-contributions-through-structu.md b/papers/items/2026-2606-22151-novelty-aware-agentic-retrieval-comparing-research-contributions-through-structu.md new file mode 100644 index 0000000..8dca78e --- /dev/null +++ b/papers/items/2026-2606-22151-novelty-aware-agentic-retrieval-comparing-research-contributions-through-structu.md @@ -0,0 +1,62 @@ +# Paper: Novelty-Aware Agentic Retrieval: Comparing Research Contributions Through Structured Multi-Step Reasoning + +--- +type: paper +title: "Novelty-Aware Agentic Retrieval: Comparing Research Contributions Through Structured Multi-Step Reasoning" +authors: Shou-Tzu Han +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22151 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-20 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, computer-use, rag, reasoning, tool-use +- arXiv categories: cs.IR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22151 diff --git a/papers/items/2026-2606-22263-revelio-cost-efficient-agentic-memory-safety-vulnerability-detection-for-reposit.md b/papers/items/2026-2606-22263-revelio-cost-efficient-agentic-memory-safety-vulnerability-detection-for-reposit.md new file mode 100644 index 0000000..f4974b9 --- /dev/null +++ b/papers/items/2026-2606-22263-revelio-cost-efficient-agentic-memory-safety-vulnerability-detection-for-reposit.md @@ -0,0 +1,64 @@ +# Paper: Revelio: Cost-Efficient Agentic Memory Safety Vulnerability Detection For Repository-Scale Codebases + +--- +type: paper +title: "Revelio: Cost-Efficient Agentic Memory Safety Vulnerability Detection For Repository-Scale Codebases" +authors: Yiwei Hou, Hao Wang, Muxi Lyu, Marius Momeu, Eric Nguyen, Taige Yang, Koushik Sen, Dawn Song, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22263 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-20 +updated_at: 2026-06-20 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.MA + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-memory, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory +- arXiv categories: cs.CR, cs.AI, cs.MA, cs.SE +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22263 diff --git a/papers/items/2026-2606-22330-hypothesis-driven-skill-optimization-for-llm-agents.md b/papers/items/2026-2606-22330-hypothesis-driven-skill-optimization-for-llm-agents.md new file mode 100644 index 0000000..a0dac4d --- /dev/null +++ b/papers/items/2026-2606-22330-hypothesis-driven-skill-optimization-for-llm-agents.md @@ -0,0 +1,64 @@ +# Paper: Hypothesis-Driven Skill Optimization for LLM Agents + +--- +type: paper +title: Hypothesis-Driven Skill Optimization for LLM Agents +authors: Fangxin Shang, Yehui Yang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22330 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-06-21 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-safety, coding-agent, memory, planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22330 diff --git a/papers/items/2026-2606-22388-planbench-xl-evaluating-long-horizon-planning-of-llm-tool-use-agents-in-large-sc.md b/papers/items/2026-2606-22388-planbench-xl-evaluating-long-horizon-planning-of-llm-tool-use-agents-in-large-sc.md new file mode 100644 index 0000000..be7dd7a --- /dev/null +++ b/papers/items/2026-2606-22388-planbench-xl-evaluating-long-horizon-planning-of-llm-tool-use-agents-in-large-sc.md @@ -0,0 +1,62 @@ +# Paper: PlanBench-XL: Evaluating Long-Horizon Planning of LLM Tool-Use Agents in Large-Scale Tool Ecosystems + +--- +type: paper +title: "PlanBench-XL: Evaluating Long-Horizon Planning of LLM Tool-Use Agents in Large-Scale Tool Ecosystems" +authors: Jiayu Liu, Qihan Lin, Cheng Qian, Rui Wang, Emre Can Acikgoz, Xiaocheng Yang, Jiateng Liu, Zhenhailong Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22388 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-06-21 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent, tool-use +- inferred topics: agent-evaluation, planning, rag, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22388 diff --git a/papers/items/2026-2606-22417-code-isn-t-memory-a-structural-codebase-index-inside-a-coding-agent.md b/papers/items/2026-2606-22417-code-isn-t-memory-a-structural-codebase-index-inside-a-coding-agent.md new file mode 100644 index 0000000..ce2c7dd --- /dev/null +++ b/papers/items/2026-2606-22417-code-isn-t-memory-a-structural-codebase-index-inside-a-coding-agent.md @@ -0,0 +1,61 @@ +# Paper: Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent + +--- +type: paper +title: "Code Isn't Memory: A Structural Codebase Index Inside a Coding Agent" +authors: Ishaan Bhola, Adithyan Krishnan, Sravanth Kurmala, Mukunda NS +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22417 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-06-21 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, memory, rag +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22417 diff --git a/papers/items/2026-2606-22484-governed-ai-assisted-engineering-graduated-human-oversight-for-agentic-code-gene.md b/papers/items/2026-2606-22484-governed-ai-assisted-engineering-graduated-human-oversight-for-agentic-code-gene.md new file mode 100644 index 0000000..be42d35 --- /dev/null +++ b/papers/items/2026-2606-22484-governed-ai-assisted-engineering-graduated-human-oversight-for-agentic-code-gene.md @@ -0,0 +1,62 @@ +# Paper: Governed AI-Assisted Engineering: Graduated Human Oversight for Agentic Code Generation in Regulated Domains + +--- +type: paper +title: "Governed AI-Assisted Engineering: Graduated Human Oversight for Agentic Code Generation in Regulated Domains" +authors: Richard Kang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22484 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, agent-safety, coding-agent, rag, workflow-agent +- arXiv categories: cs.HC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22484 diff --git a/papers/items/2026-2606-22495-grounded-scaling-why-agentic-ai-needs-deterministic-environments.md b/papers/items/2026-2606-22495-grounded-scaling-why-agentic-ai-needs-deterministic-environments.md new file mode 100644 index 0000000..15f8828 --- /dev/null +++ b/papers/items/2026-2606-22495-grounded-scaling-why-agentic-ai-needs-deterministic-environments.md @@ -0,0 +1,62 @@ +# Paper: Grounded Scaling: Why Agentic AI Needs Deterministic Environments + +--- +type: paper +title: "Grounded Scaling: Why Agentic AI Needs Deterministic Environments" +authors: Liang Ding, Xintong Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22495 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-06-21 +status: queued +relevance: high +topics: + - agent-safety + - embodied-agent + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-safety, embodied-agent, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22495 diff --git a/papers/items/2026-2606-22557-macagentbench-benchmarking-ai-agents-on-real-world-macos-desktop.md b/papers/items/2026-2606-22557-macagentbench-benchmarking-ai-agents-on-real-world-macos-desktop.md new file mode 100644 index 0000000..750ceef --- /dev/null +++ b/papers/items/2026-2606-22557-macagentbench-benchmarking-ai-agents-on-real-world-macos-desktop.md @@ -0,0 +1,65 @@ +# Paper: MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop + +--- +type: paper +title: "MacAgentBench: Benchmarking AI Agents on Real-World macOS Desktop" +authors: Yikun Fu, Bowen Fu, Zhenyu Wu, Shuang Cheng, Xiaowei Sun, Bowen Yang, Zehao Li, Yibo Zhao, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22557 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-06-21 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: agent-evaluation, ai-agent, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, ai-agent, web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CL, cs.HC +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22557 diff --git a/papers/items/2026-2606-22610-paperclaw-harnessing-agents-for-autonomous-research-and-human-in-the-loop-refine.md b/papers/items/2026-2606-22610-paperclaw-harnessing-agents-for-autonomous-research-and-human-in-the-loop-refine.md new file mode 100644 index 0000000..34c7d02 --- /dev/null +++ b/papers/items/2026-2606-22610-paperclaw-harnessing-agents-for-autonomous-research-and-human-in-the-loop-refine.md @@ -0,0 +1,61 @@ +# Paper: PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement + +--- +type: paper +title: "PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement" +authors: Weiwei Ye, Hangchen Liu, Dongyuan Li, Renhe Jiang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22610 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-06-21 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, memory, multi-agent, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22610 diff --git a/papers/items/2026-2606-22647-raven-agentic-rag-for-automated-vulnerability-repair.md b/papers/items/2026-2606-22647-raven-agentic-rag-for-automated-vulnerability-repair.md new file mode 100644 index 0000000..2851732 --- /dev/null +++ b/papers/items/2026-2606-22647-raven-agentic-rag-for-automated-vulnerability-repair.md @@ -0,0 +1,64 @@ +# Paper: RAVEN: Agentic RAG for Automated Vulnerability Repair + +--- +type: paper +title: "RAVEN: Agentic RAG for Automated Vulnerability Repair" +authors: Varun Gadey, Zijie Liu, Alexandra Dmitrienko +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22647 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-06-21 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - memory + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.LG + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, memory, rag +- arXiv categories: cs.CR, cs.LG, cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22647 diff --git a/papers/items/2026-2606-22673-agentlens-interpretable-safety-steering-via-mechanistic-subspaces-for-multi-turn.md b/papers/items/2026-2606-22673-agentlens-interpretable-safety-steering-via-mechanistic-subspaces-for-multi-turn.md new file mode 100644 index 0000000..30b8dd4 --- /dev/null +++ b/papers/items/2026-2606-22673-agentlens-interpretable-safety-steering-via-mechanistic-subspaces-for-multi-turn.md @@ -0,0 +1,62 @@ +# Paper: AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent + +--- +type: paper +title: "AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent" +authors: Weidi Luo, Qiming Zhang, Yihao Quan, Mingyu Jin, Jie Cai, Chaowei Xiao, Jingcheng Niu, Zhen Xiang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22673 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-06-21 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-safety, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.AI, cs.SE +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22673 diff --git a/papers/items/2026-2606-22678-rigorbench-benchmarking-engineering-process-discipline-in-autonomous-ai-coding-a.md b/papers/items/2026-2606-22678-rigorbench-benchmarking-engineering-process-discipline-in-autonomous-ai-coding-a.md new file mode 100644 index 0000000..2b58c07 --- /dev/null +++ b/papers/items/2026-2606-22678-rigorbench-benchmarking-engineering-process-discipline-in-autonomous-ai-coding-a.md @@ -0,0 +1,63 @@ +# Paper: RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents + +--- +type: paper +title: "RigorBench: Benchmarking Engineering Process Discipline in Autonomous AI Coding Agents" +authors: Meher Bhaskar Madiraju, Meher Sai Preetam Madiraju +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22678 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-21 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, planning, rag, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22678 diff --git a/papers/items/2026-2606-22737-groundeval-a-deterministic-replacement-for-llm-as-judge-in-stateful-agent-evalua.md b/papers/items/2026-2606-22737-groundeval-a-deterministic-replacement-for-llm-as-judge-in-stateful-agent-evalua.md new file mode 100644 index 0000000..454cf04 --- /dev/null +++ b/papers/items/2026-2606-22737-groundeval-a-deterministic-replacement-for-llm-as-judge-in-stateful-agent-evalua.md @@ -0,0 +1,62 @@ +# Paper: GroundEval: A Deterministic Replacement for LLM-as-Judge in Stateful Agent Evaluation + +--- +type: paper +title: "GroundEval: A Deterministic Replacement for LLM-as-Judge in Stateful Agent Evaluation" +authors: Jeffrey Flynt +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22737 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, memory, tool-use +- arXiv categories: cs.AI, cs.CL, cs.SE +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22737 diff --git a/papers/items/2026-2606-22741-grade-graph-representation-of-llm-agent-dependency-and-execution.md b/papers/items/2026-2606-22741-grade-graph-representation-of-llm-agent-dependency-and-execution.md new file mode 100644 index 0000000..df23057 --- /dev/null +++ b/papers/items/2026-2606-22741-grade-graph-representation-of-llm-agent-dependency-and-execution.md @@ -0,0 +1,60 @@ +# Paper: GRADE: Graph Representation of LLM Agent Dependency and Execution + +--- +type: paper +title: "GRADE: Graph Representation of LLM Agent Dependency and Execution" +authors: Yue Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22741 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - coding-agent + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm, tool-use +- inferred topics: coding-agent, multi-agent, tool-use +- arXiv categories: cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22741 diff --git a/papers/items/2026-2606-22844-ramem-contextual-reinstatement-for-long-term-agentic-memory.md b/papers/items/2026-2606-22844-ramem-contextual-reinstatement-for-long-term-agentic-memory.md new file mode 100644 index 0000000..18ebd27 --- /dev/null +++ b/papers/items/2026-2606-22844-ramem-contextual-reinstatement-for-long-term-agentic-memory.md @@ -0,0 +1,62 @@ +# Paper: RaMem: Contextual Reinstatement for Long-term Agentic Memory + +--- +type: paper +title: "RaMem: Contextual Reinstatement for Long-term Agentic Memory" +authors: Wei Yang, Bryce Kan, Shixuan Li, Li Li, Yuehan Qin, Jiate Li, Paul Bogdan, Jesse Thomason +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22844 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.AI, cs.MA +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22844 diff --git a/papers/items/2026-2606-22864-when-auc-0-998-is-not-enough-a-candidate-evaluation-protocol-for-hidden-state-pr.md b/papers/items/2026-2606-22864-when-auc-0-998-is-not-enough-a-candidate-evaluation-protocol-for-hidden-state-pr.md new file mode 100644 index 0000000..0cacb8c --- /dev/null +++ b/papers/items/2026-2606-22864-when-auc-0-998-is-not-enough-a-candidate-evaluation-protocol-for-hidden-state-pr.md @@ -0,0 +1,60 @@ +# Paper: When AUC 0.998 Is Not Enough: A Candidate Evaluation Protocol for Hidden-State Probes of Indirect Prompt Injection in Multimodal Computer-Use Agents + +--- +type: paper +title: "When AUC 0.998 Is Not Enough: A Candidate Evaluation Protocol for Hidden-State Probes of Indirect Prompt Injection in Multimodal Computer-Use Agents" +authors: Yanhang Li, Zhichao Fan, Zexin Zhuang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22864 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22864 diff --git a/papers/items/2026-2606-22948-envs-environment-native-verified-search-for-long-horizon-gui-agents.md b/papers/items/2026-2606-22948-envs-environment-native-verified-search-for-long-horizon-gui-agents.md new file mode 100644 index 0000000..928e63c --- /dev/null +++ b/papers/items/2026-2606-22948-envs-environment-native-verified-search-for-long-horizon-gui-agents.md @@ -0,0 +1,63 @@ +# Paper: ENVS: Environment-Native Verified Search for Long-Horizon GUI Agents + +--- +type: paper +title: "ENVS: Environment-Native Verified Search for Long-Horizon GUI Agents" +authors: Yincheng Zhou, Athena Zhuoming Zhong, Shijie Zhang, Kevin Zhang, Teresa Xiaotao Shang, Shanghang Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22948 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.CV +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22948 diff --git a/papers/items/2026-2606-22953-plans-don-t-persist-why-context-management-is-load-bearing-for-llm-agents.md b/papers/items/2026-2606-22953-plans-don-t-persist-why-context-management-is-load-bearing-for-llm-agents.md new file mode 100644 index 0000000..158f054 --- /dev/null +++ b/papers/items/2026-2606-22953-plans-don-t-persist-why-context-management-is-load-bearing-for-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: Plans Don't Persist: Why Context Management Is Load Bearing for LLM Agents + +--- +type: paper +title: "Plans Don't Persist: Why Context Management Is Load Bearing for LLM Agents" +authors: Aman Mehta, Anupam Datta +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.22953 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.22953 diff --git a/papers/items/2026-2606-23032-ipo-finance-agent-benchmark-of-llm-financial-analysts-beyond-finance-agent-v2-wi.md b/papers/items/2026-2606-23032-ipo-finance-agent-benchmark-of-llm-financial-analysts-beyond-finance-agent-v2-wi.md new file mode 100644 index 0000000..1cd9500 --- /dev/null +++ b/papers/items/2026-2606-23032-ipo-finance-agent-benchmark-of-llm-financial-analysts-beyond-finance-agent-v2-wi.md @@ -0,0 +1,62 @@ +# Paper: IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO + +--- +type: paper +title: "IPO Finance Agent: Benchmark of LLM Financial Analysts Beyond Finance Agent v2, with Automated Rubric Generation, on the SpaceX (SPCX) IPO" +authors: Mostapha Benhenda +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23032 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - q-fin.GN +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.AI, q-fin.GN +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23032 diff --git a/papers/items/2026-2606-23130-understanding-the-in-security-of-vibe-coded-applications.md b/papers/items/2026-2606-23130-understanding-the-in-security-of-vibe-coded-applications.md new file mode 100644 index 0000000..8dcf46f --- /dev/null +++ b/papers/items/2026-2606-23130-understanding-the-in-security-of-vibe-coded-applications.md @@ -0,0 +1,64 @@ +# Paper: Understanding the (In)Security of Vibe-Coded Applications + +--- +type: paper +title: Understanding the (In)Security of Vibe-Coded Applications +authors: Junquan Deng, Zhiyu Fan, Ruijie Meng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23130 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.SE +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23130 diff --git a/papers/items/2026-2606-23195-memory-contagion-cross-temporal-propagation-of-evaluator-bias-via-agent-memory.md b/papers/items/2026-2606-23195-memory-contagion-cross-temporal-propagation-of-evaluator-bias-via-agent-memory.md new file mode 100644 index 0000000..25c6dd3 --- /dev/null +++ b/papers/items/2026-2606-23195-memory-contagion-cross-temporal-propagation-of-evaluator-bias-via-agent-memory.md @@ -0,0 +1,63 @@ +# Paper: Memory Contagion: Cross-Temporal Propagation of Evaluator Bias via Agent Memory + +--- +type: paper +title: "Memory Contagion: Cross-Temporal Propagation of Evaluator Bias via Agent Memory" +authors: Zewen Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23195 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, tool-use +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23195 diff --git a/papers/items/2026-2606-23283-towards-root-memories-benchmarking-and-enhancing-implicit-logical-memory-retriev.md b/papers/items/2026-2606-23283-towards-root-memories-benchmarking-and-enhancing-implicit-logical-memory-retriev.md new file mode 100644 index 0000000..35b7bda --- /dev/null +++ b/papers/items/2026-2606-23283-towards-root-memories-benchmarking-and-enhancing-implicit-logical-memory-retriev.md @@ -0,0 +1,60 @@ +# Paper: Towards Root Memories: Benchmarking and Enhancing Implicit Logical Memory Retrieval for Personalized LLMs + +--- +type: paper +title: "Towards Root Memories: Benchmarking and Enhancing Implicit Logical Memory Retrieval for Personalized LLMs" +authors: Hongxun Ding, Xiang Yu, Chengbing Wang, Jianfei Xiao, Keqin Bao, Wenjie Wang, Xiangnan He +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23283 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23283 diff --git a/papers/items/2026-2606-23327-videoagent-all-in-one-framework-for-video-understanding-and-editing.md b/papers/items/2026-2606-23327-videoagent-all-in-one-framework-for-video-understanding-and-editing.md new file mode 100644 index 0000000..deba0b3 --- /dev/null +++ b/papers/items/2026-2606-23327-videoagent-all-in-one-framework-for-video-understanding-and-editing.md @@ -0,0 +1,63 @@ +# Paper: VideoAgent: All-in-One Framework for Video Understanding and Editing + +--- +type: paper +title: "VideoAgent: All-in-One Framework for Video Understanding and Editing" +authors: Hengji Zhou, Lingxuan Huang, Jian Wang, Bing Zhou, Si Wu, Lianghao Xia, Chao Huang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23327 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, planning, rag, tool-use +- arXiv categories: cs.CV, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23327 diff --git a/papers/items/2026-2606-23343-group-selection-promotes-prosocial-prompts-in-populations-of-llm-agents.md b/papers/items/2026-2606-23343-group-selection-promotes-prosocial-prompts-in-populations-of-llm-agents.md new file mode 100644 index 0000000..071a7ea --- /dev/null +++ b/papers/items/2026-2606-23343-group-selection-promotes-prosocial-prompts-in-populations-of-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: Group Selection Promotes Prosocial Prompts in Populations of LLM Agents + +--- +type: paper +title: Group Selection Promotes Prosocial Prompts in Populations of LLM Agents +authors: Luis Celiktemel, Edward Eichhorn, Levin Brinkmann, Robin Schimmelpfennig, Aron Vallinder, Yaomin Jiang, Edward Hughes, Iyad Rahwan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23343 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - coding-agent + - computer-use + - multi-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: coding-agent, computer-use, multi-agent, world-model +- arXiv categories: cs.CY +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23343 diff --git a/papers/items/2026-2606-23565-holoagent-0-a-unified-embodied-agent-framework-with-3d-spatial-memory.md b/papers/items/2026-2606-23565-holoagent-0-a-unified-embodied-agent-framework-with-3d-spatial-memory.md new file mode 100644 index 0000000..60cd719 --- /dev/null +++ b/papers/items/2026-2606-23565-holoagent-0-a-unified-embodied-agent-framework-with-3d-spatial-memory.md @@ -0,0 +1,66 @@ +# Paper: HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory + +--- +type: paper +title: "HoloAgent-0: A Unified Embodied Agent Framework with 3D Spatial Memory" +authors: Xiaolin Zhou, Liu Liu, Tingyang Xiao, Wei Feng, Fa Fu, Xinrui Meng, Xinjie Wang, Jialiang Han, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23565 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - embodied-agent + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, embodied-agent, memory, planning, rag, tool-use +- arXiv categories: cs.RO, cs.CV +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23565 diff --git a/papers/items/2026-2606-23664-mas-promptbench-when-does-prompt-optimization-improve-multi-agent-llm-systems.md b/papers/items/2026-2606-23664-mas-promptbench-when-does-prompt-optimization-improve-multi-agent-llm-systems.md new file mode 100644 index 0000000..b643681 --- /dev/null +++ b/papers/items/2026-2606-23664-mas-promptbench-when-does-prompt-optimization-improve-multi-agent-llm-systems.md @@ -0,0 +1,61 @@ +# Paper: MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems? + +--- +type: paper +title: "MAS-PromptBench: When Does Prompt Optimization Improve Multi-Agent LLM Systems?" +authors: Juyang Bai, Laixi Shi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23664 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, workflow-agent +- arXiv categories: cs.LG, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23664 diff --git a/papers/items/2026-2606-23752-esaa-conversational-an-event-sourced-memory-layer-for-continuity-handoff-and-cur.md b/papers/items/2026-2606-23752-esaa-conversational-an-event-sourced-memory-layer-for-continuity-handoff-and-cur.md new file mode 100644 index 0000000..691dd27 --- /dev/null +++ b/papers/items/2026-2606-23752-esaa-conversational-an-event-sourced-memory-layer-for-continuity-handoff-and-cur.md @@ -0,0 +1,60 @@ +# Paper: ESAA-Conversational: An Event-Sourced Memory Layer for Continuity, Handoff, and Curation Across Heterogeneous LLM Coding Agents + +--- +type: paper +title: "ESAA-Conversational: An Event-Sourced Memory Layer for Continuity, Handoff, and Curation Across Heterogeneous LLM Coding Agents" +authors: Elzo Brito dos Santos Filho +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23752 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - coding-agent + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: coding-agent, memory, tool-use +- arXiv categories: cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23752 diff --git a/papers/items/2026-2606-23764-emergent-relational-order-in-llm-agent-societies-from-collective-affect-to-autho.md b/papers/items/2026-2606-23764-emergent-relational-order-in-llm-agent-societies-from-collective-affect-to-autho.md new file mode 100644 index 0000000..36c0450 --- /dev/null +++ b/papers/items/2026-2606-23764-emergent-relational-order-in-llm-agent-societies-from-collective-affect-to-autho.md @@ -0,0 +1,62 @@ +# Paper: Emergent Relational Order in LLM Agent Societies: From Collective Affect to Authority Stratification + +--- +type: paper +title: "Emergent Relational Order in LLM Agent Societies: From Collective Affect to Authority Stratification" +authors: Zhiyuan Ji, Xinyu Chen, Ziqi Dai, Shiyun Tang, Chunyu Wei, Yueguo Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23764 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - multi-agent + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: multi-agent, planning, tool-use, world-model +- arXiv categories: cs.MA, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23764 diff --git a/papers/items/2026-2606-23927-rift-bench-dynamic-red-teaming-for-agentic-ai-systems.md b/papers/items/2026-2606-23927-rift-bench-dynamic-red-teaming-for-agentic-ai-systems.md new file mode 100644 index 0000000..062c0e3 --- /dev/null +++ b/papers/items/2026-2606-23927-rift-bench-dynamic-red-teaming-for-agentic-ai-systems.md @@ -0,0 +1,61 @@ +# Paper: RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems + +--- +type: paper +title: "RIFT-Bench: Dynamic Red-teaming For Agentic AI Systems" +authors: Yarin Yerushalmi Levi, Roy Betser, Amit Giloni, Lidor Erez, Itay Gershon, Oren Rachmil, Sindhu Padakandla, Roman Vainshtein +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23927 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23927 diff --git a/papers/items/2026-2606-23991-critique-of-agent-model.md b/papers/items/2026-2606-23991-critique-of-agent-model.md new file mode 100644 index 0000000..e523e10 --- /dev/null +++ b/papers/items/2026-2606-23991-critique-of-agent-model.md @@ -0,0 +1,67 @@ +# Paper: Critique of Agent Model + +--- +type: paper +title: Critique of Agent Model +authors: Eric Xing, Mingkai Deng, Jinyu Hou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.23991 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - rag + - reasoning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG + - cs.MA + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agentic-ai, ai-agent, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, ai-agent, coding-agent +- inferred topics: agent-safety, coding-agent, rag, reasoning, tool-use, workflow-agent, world-model +- arXiv categories: cs.AI, cs.LG, cs.MA, cs.RO +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.23991 diff --git a/papers/items/2026-2606-24193-skychain-intelligence-a-blockchain-secured-multi-agent-drl-framework-for-low-alt.md b/papers/items/2026-2606-24193-skychain-intelligence-a-blockchain-secured-multi-agent-drl-framework-for-low-alt.md new file mode 100644 index 0000000..3cea856 --- /dev/null +++ b/papers/items/2026-2606-24193-skychain-intelligence-a-blockchain-secured-multi-agent-drl-framework-for-low-alt.md @@ -0,0 +1,63 @@ +# Paper: SkyChain Intelligence: A Blockchain-Secured Multi-Agent DRL Framework for Low-Altitude Embodied Artificial Intelligence + +--- +type: paper +title: "SkyChain Intelligence: A Blockchain-Secured Multi-Agent DRL Framework for Low-Altitude Embodied Artificial Intelligence" +authors: Haoxiang Luo, Tianqi Jiang, Ruichen Zhang, Yinqiu Liu, Gang Sun, Hongfang Yu, Abbas Jamalipour, Dong In Kim +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24193 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-safety + - embodied-agent + - multi-agent + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.NI + - cs.DC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-safety, embodied-agent, multi-agent, tool-use, world-model +- arXiv categories: cs.NI, cs.DC +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24193 diff --git a/papers/items/2026-2606-24235-sp-mind-an-autonomous-reasoning-agent-for-spatial-proteomics-analysis.md b/papers/items/2026-2606-24235-sp-mind-an-autonomous-reasoning-agent-for-spatial-proteomics-analysis.md new file mode 100644 index 0000000..838cc4a --- /dev/null +++ b/papers/items/2026-2606-24235-sp-mind-an-autonomous-reasoning-agent-for-spatial-proteomics-analysis.md @@ -0,0 +1,63 @@ +# Paper: SP-Mind: An Autonomous Reasoning Agent for Spatial Proteomics Analysis + +--- +type: paper +title: "SP-Mind: An Autonomous Reasoning Agent for Spatial Proteomics Analysis" +authors: Yucheng Yuan, Yuanfeng Ji, Zhongxiao Li, Ruijiang Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24235 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, computer-use, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24235 diff --git a/papers/items/2026-2606-24322-securing-llm-agent-long-term-memory-against-poisoning-non-malleable-origin-bound.md b/papers/items/2026-2606-24322-securing-llm-agent-long-term-memory-against-poisoning-non-malleable-origin-bound.md new file mode 100644 index 0000000..3e5e6c7 --- /dev/null +++ b/papers/items/2026-2606-24322-securing-llm-agent-long-term-memory-against-poisoning-non-malleable-origin-bound.md @@ -0,0 +1,60 @@ +# Paper: Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees + +--- +type: paper +title: "Securing LLM-Agent Long-Term Memory Against Poisoning: Non-Malleable, Origin-Bound Authority with Machine-Checked Guarantees" +authors: Yedidel Louck +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24322 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, tool-use +- arXiv categories: cs.CR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24322 diff --git a/papers/items/2026-2606-24402-poisoned-playbooks-demystifying-knowledge-poisoning-effects-on-ai-security-agent.md b/papers/items/2026-2606-24402-poisoned-playbooks-demystifying-knowledge-poisoning-effects-on-ai-security-agent.md new file mode 100644 index 0000000..ad8751f --- /dev/null +++ b/papers/items/2026-2606-24402-poisoned-playbooks-demystifying-knowledge-poisoning-effects-on-ai-security-agent.md @@ -0,0 +1,62 @@ +# Paper: Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents + +--- +type: paper +title: "Poisoned Playbooks: Demystifying Knowledge Poisoning Effects on AI Security Agents" +authors: Juho Park, Hyunmin Choi, Kevin Nam +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24402 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: ai-agent, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, rag-agent +- inferred topics: agent-evaluation, agent-safety, rag, reasoning, tool-use +- arXiv categories: cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24402 diff --git a/papers/items/2026-2606-24416-agentic-ai-for-bilevel-long-term-optimization-of-policy-driven-physical-layer-sy.md b/papers/items/2026-2606-24416-agentic-ai-for-bilevel-long-term-optimization-of-policy-driven-physical-layer-sy.md new file mode 100644 index 0000000..0f019aa --- /dev/null +++ b/papers/items/2026-2606-24416-agentic-ai-for-bilevel-long-term-optimization-of-policy-driven-physical-layer-sy.md @@ -0,0 +1,61 @@ +# Paper: Agentic AI for Bilevel Long-Term Optimization of Policy-Driven Physical Layer Systems + +--- +type: paper +title: Agentic AI for Bilevel Long-Term Optimization of Policy-Driven Physical Layer Systems +authors: Bingnan Xiao, Chenhao Yang, Wei Ni, Xin Wang, Tony Q. S. Quek +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24416 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, memory, multi-agent, rag +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24416 diff --git a/papers/items/2026-2606-24437-rem-moa-reasoning-memory-sustains-mixture-of-agents-scaling.md b/papers/items/2026-2606-24437-rem-moa-reasoning-memory-sustains-mixture-of-agents-scaling.md new file mode 100644 index 0000000..68fc29a --- /dev/null +++ b/papers/items/2026-2606-24437-rem-moa-reasoning-memory-sustains-mixture-of-agents-scaling.md @@ -0,0 +1,61 @@ +# Paper: ReM-MoA: Reasoning Memory Sustains Mixture-of-Agents Scaling + +--- +type: paper +title: "ReM-MoA: Reasoning Memory Sustains Mixture-of-Agents Scaling" +authors: Heng Ping, Arijit Bhattacharjee, Peiyu Zhang, Shixuan Li, Wei Yang, Ali Jannesari, Nesreen Ahmed, Paul Bogdan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24437 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, memory, multi-agent, reasoning +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24437 diff --git a/papers/items/2026-2606-24453-bayesian-control-for-coding-agents.md b/papers/items/2026-2606-24453-bayesian-control-for-coding-agents.md new file mode 100644 index 0000000..bebf1c6 --- /dev/null +++ b/papers/items/2026-2606-24453-bayesian-control-for-coding-agents.md @@ -0,0 +1,62 @@ +# Paper: Bayesian control for coding agents + +--- +type: paper +title: Bayesian control for coding agents +authors: Theodore Papamarkou, Vladislav Smirnov, Viktor Mazanov, Artem Vazhentsev, Preslav Nakov, Timothy Baldwin, Artem Shelmanov +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24453 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: coding-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, tool-use +- inferred topics: agent-evaluation, coding-agent, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24453 diff --git a/papers/items/2026-2606-24515-reinforcement-learning-for-computer-use-agents-with-autonomous-evaluation.md b/papers/items/2026-2606-24515-reinforcement-learning-for-computer-use-agents-with-autonomous-evaluation.md new file mode 100644 index 0000000..9b7345e --- /dev/null +++ b/papers/items/2026-2606-24515-reinforcement-learning-for-computer-use-agents-with-autonomous-evaluation.md @@ -0,0 +1,61 @@ +# Paper: Reinforcement Learning for Computer-Use Agents with Autonomous Evaluation + +--- +type: paper +title: Reinforcement Learning for Computer-Use Agents with Autonomous Evaluation +authors: Marta Sumyk, Oleksandr Kosovan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24515 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, rag +- arXiv categories: cs.AI, cs.HC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24515 diff --git a/papers/items/2026-2606-24525-viscritic-visual-state-comparison-as-process-reward-for-gui-agents.md b/papers/items/2026-2606-24525-viscritic-visual-state-comparison-as-process-reward-for-gui-agents.md new file mode 100644 index 0000000..067a61f --- /dev/null +++ b/papers/items/2026-2606-24525-viscritic-visual-state-comparison-as-process-reward-for-gui-agents.md @@ -0,0 +1,62 @@ +# Paper: VisCritic: Visual State Comparison as Process Reward for GUI Agents + +--- +type: paper +title: "VisCritic: Visual State Comparison as Process Reward for GUI Agents" +authors: Jiachen Qian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24525 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, reasoning, tool-use +- arXiv categories: cs.CV +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24525 diff --git a/papers/items/2026-2606-24535-governed-shared-memory-for-multi-agent-llm-systems.md b/papers/items/2026-2606-24535-governed-shared-memory-for-multi-agent-llm-systems.md new file mode 100644 index 0000000..aca6e6c --- /dev/null +++ b/papers/items/2026-2606-24535-governed-shared-memory-for-multi-agent-llm-systems.md @@ -0,0 +1,62 @@ +# Paper: Governed Shared Memory for Multi-Agent LLM Systems + +--- +type: paper +title: Governed Shared Memory for Multi-Agent LLM Systems +authors: Yanki Margalit, Nurit Cohen-Inger, Erni Avram, Ran Taig, Oded Margalit +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24535 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-memory, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, multi-agent-llm +- inferred topics: agent-evaluation, memory, multi-agent, rag, tool-use +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24535 diff --git a/papers/items/2026-2606-24551-gui-vs-cli-execution-bottlenecks-in-screen-only-and-skill-mediated-computer-use-.md b/papers/items/2026-2606-24551-gui-vs-cli-execution-bottlenecks-in-screen-only-and-skill-mediated-computer-use-.md new file mode 100644 index 0000000..99c6305 --- /dev/null +++ b/papers/items/2026-2606-24551-gui-vs-cli-execution-bottlenecks-in-screen-only-and-skill-mediated-computer-use-.md @@ -0,0 +1,64 @@ +# Paper: GUI vs. CLI: Execution Bottlenecks in Screen-Only and Skill-Mediated Computer-Use Agents + +--- +type: paper +title: "GUI vs. CLI: Execution Bottlenecks in Screen-Only and Skill-Mediated Computer-Use Agents" +authors: Xiao Zhou, Siyue Zhang, Yilun Zhao, Jinbiao Wei, Tingyu Song, Arman Cohan, Chen Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24551 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24551 diff --git a/papers/items/2026-2606-24595-memprobe-probing-long-term-agent-memory-via-hidden-user-state-recovery.md b/papers/items/2026-2606-24595-memprobe-probing-long-term-agent-memory-via-hidden-user-state-recovery.md new file mode 100644 index 0000000..47cc88c --- /dev/null +++ b/papers/items/2026-2606-24595-memprobe-probing-long-term-agent-memory-via-hidden-user-state-recovery.md @@ -0,0 +1,61 @@ +# Paper: MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery + +--- +type: paper +title: "MEMPROBE: Probing Long-Term Agent Memory via Hidden User-State Recovery" +authors: Enze Ma, Yufan Zhou, Wei-Chieh Huang, Jie Yang, Huanhuan Ma, Zixuan Wang, Chengze Li, Chunyu Miao, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24595 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24595 diff --git a/papers/items/2026-2606-24597-qwen-agentworld-language-world-models-for-general-agents.md b/papers/items/2026-2606-24597-qwen-agentworld-language-world-models-for-general-agents.md new file mode 100644 index 0000000..9ee00fe --- /dev/null +++ b/papers/items/2026-2606-24597-qwen-agentworld-language-world-models-for-general-agents.md @@ -0,0 +1,63 @@ +# Paper: Qwen-AgentWorld: Language World Models for General Agents + +--- +type: paper +title: "Qwen-AgentWorld: Language World Models for General Agents" +authors: Yuxin Zuo, Zikai Xiao, Li Sheng, Fei Huang, Jianhong Tu, Yuxuan Liu, Tianyi Tang, Xiaomeng Hu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24597 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use, world-model +- arXiv categories: cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24597 diff --git a/papers/items/2026-2606-24623-privacy-preserving-rag-via-multi-agent-semantic-rewriting-achieving-confidential.md b/papers/items/2026-2606-24623-privacy-preserving-rag-via-multi-agent-semantic-rewriting-achieving-confidential.md new file mode 100644 index 0000000..e3f9dec --- /dev/null +++ b/papers/items/2026-2606-24623-privacy-preserving-rag-via-multi-agent-semantic-rewriting-achieving-confidential.md @@ -0,0 +1,63 @@ +# Paper: Privacy-Preserving RAG via Multi-Agent Semantic Rewriting: Achieving Confidentiality Without Compromising Contextual Fidelity + +--- +type: paper +title: "Privacy-Preserving RAG via Multi-Agent Semantic Rewriting: Achieving Confidentiality Without Compromising Contextual Fidelity" +authors: Yuanhe Zhao, Tianyu Zhang, Huafei Xing, Derek F. Wong, Jianbin Li, Tao Fang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24623 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm, rag-agent +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24623 diff --git a/papers/items/2026-2606-24626-safari-scaling-long-horizon-agentic-fault-attribution-via-active-investigation.md b/papers/items/2026-2606-24626-safari-scaling-long-horizon-agentic-fault-attribution-via-active-investigation.md new file mode 100644 index 0000000..ecc8034 --- /dev/null +++ b/papers/items/2026-2606-24626-safari-scaling-long-horizon-agentic-fault-attribution-via-active-investigation.md @@ -0,0 +1,63 @@ +# Paper: SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation + +--- +type: paper +title: "SAFARI: Scaling Long Horizon Agentic Fault Attribution via Active Investigation" +authors: Chenyang Zhu, Jiayu Yao, Kushal Chawla, Youbing Yin, Nathan Wolfe, Pengshan Cai, Jingyu Wu, Spencer Hong, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24626 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: autonomous-agent-llm, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, multi-agent-llm +- inferred topics: agent-evaluation, memory, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24626 diff --git a/papers/items/2026-2606-24649-agentic-collaborative-cognition-for-zero-shot-3d-understanding.md b/papers/items/2026-2606-24649-agentic-collaborative-cognition-for-zero-shot-3d-understanding.md new file mode 100644 index 0000000..570d919 --- /dev/null +++ b/papers/items/2026-2606-24649-agentic-collaborative-cognition-for-zero-shot-3d-understanding.md @@ -0,0 +1,62 @@ +# Paper: Agentic Collaborative Cognition for Zero-Shot 3D Understanding + +--- +type: paper +title: Agentic Collaborative Cognition for Zero-Shot 3D Understanding +authors: Wenxin Wang, Bo Zhang, Feng Chen, Zixuan Wang, Wen Li, Changsheng Li, Yinjie Lei +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24649 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, computer-use, multi-agent, planning, rag +- arXiv categories: cs.CV +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24649 diff --git a/papers/items/2026-2606-24689-automated-summarization-of-software-documents-an-llm-based-multi-agent-approach.md b/papers/items/2026-2606-24689-automated-summarization-of-software-documents-an-llm-based-multi-agent-approach.md new file mode 100644 index 0000000..565c014 --- /dev/null +++ b/papers/items/2026-2606-24689-automated-summarization-of-software-documents-an-llm-based-multi-agent-approach.md @@ -0,0 +1,61 @@ +# Paper: Automated Summarization of Software Documents: An LLM-based Multi-Agent Approach + +--- +type: paper +title: "Automated Summarization of Software Documents: An LLM-based Multi-Agent Approach" +authors: Duc S. H. Nguyen, Minh T. Nguyen, Phuong T. Nguyen, Juri Di Rocco, Davide Di Ruscio +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24689 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent, workflow-agent +- arXiv categories: cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24689 diff --git a/papers/items/2026-2606-24694-supplynet-supporting-visual-exploratory-learning-in-supply-chain-via-contextual-.md b/papers/items/2026-2606-24694-supplynet-supporting-visual-exploratory-learning-in-supply-chain-via-contextual-.md new file mode 100644 index 0000000..f1f5e4d --- /dev/null +++ b/papers/items/2026-2606-24694-supplynet-supporting-visual-exploratory-learning-in-supply-chain-via-contextual-.md @@ -0,0 +1,61 @@ +# Paper: SupplyNet: Supporting Visual Exploratory Learning in Supply Chain via Contextual Multi-Agent Simulation + +--- +type: paper +title: "SupplyNet: Supporting Visual Exploratory Learning in Supply Chain via Contextual Multi-Agent Simulation" +authors: Yanjia Li, Kelcy Kexin Han, Tianrui Hu, Yi-Fan Cao, Huamin Qu, Sicheng Song +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24694 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - multi-agent + - rag + - reasoning + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: multi-agent, rag, reasoning, world-model +- arXiv categories: cs.HC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24694 diff --git a/papers/items/2026-2606-24775-are-we-ready-for-an-agent-native-memory-system.md b/papers/items/2026-2606-24775-are-we-ready-for-an-agent-native-memory-system.md new file mode 100644 index 0000000..4905484 --- /dev/null +++ b/papers/items/2026-2606-24775-are-we-ready-for-an-agent-native-memory-system.md @@ -0,0 +1,64 @@ +# Paper: Are We Ready For An Agent-Native Memory System? + +--- +type: paper +title: Are We Ready For An Agent-Native Memory System? +authors: Wei Zhou, Xuanhe Zhou, Shaokun Han, Hongming Xu, Guoliang Li, Zhiyu Li, Feiyu Xiong, Fan Wu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24775 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.DB + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag, tool-use +- arXiv categories: cs.CL, cs.DB, cs.IR +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24775 diff --git a/papers/items/2026-2606-24779-deepbd-a-grounded-agentic-workflow-for-variant-prioritization-and-diagnosis-of-g.md b/papers/items/2026-2606-24779-deepbd-a-grounded-agentic-workflow-for-variant-prioritization-and-diagnosis-of-g.md new file mode 100644 index 0000000..91ac291 --- /dev/null +++ b/papers/items/2026-2606-24779-deepbd-a-grounded-agentic-workflow-for-variant-prioritization-and-diagnosis-of-g.md @@ -0,0 +1,62 @@ +# Paper: DeepBD: A Grounded Agentic Workflow for Variant Prioritization and Diagnosis of Genetic Birth Defects + +--- +type: paper +title: "DeepBD: A Grounded Agentic Workflow for Variant Prioritization and Diagnosis of Genetic Birth Defects" +authors: Shiyu Li, Ziqi Yan, Zhihao Wu, Jielong Lu, Weiran Liao, Jiajun Yu, Genjie Li, Zeyu Chu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24779 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - q-bio.GN + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, memory, tool-use, workflow-agent +- arXiv categories: q-bio.GN, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24779 diff --git a/papers/items/2026-2606-24820-sherloc-structured-diagnostic-localization-for-code-repair-agents.md b/papers/items/2026-2606-24820-sherloc-structured-diagnostic-localization-for-code-repair-agents.md new file mode 100644 index 0000000..dfdceb5 --- /dev/null +++ b/papers/items/2026-2606-24820-sherloc-structured-diagnostic-localization-for-code-repair-agents.md @@ -0,0 +1,63 @@ +# Paper: SHERLOC: Structured Diagnostic Localization for Code Repair Agents + +--- +type: paper +title: "SHERLOC: Structured Diagnostic Localization for Code Repair Agents" +authors: Hovhannes Tamoyan, Sean Narenthiran, Erik Arakelyan, Mira Mezini, Boris Ginsburg +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24820 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: coding-agent, multi-agent-llm, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, multi-agent-llm, tool-use +- inferred topics: agent-evaluation, coding-agent, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24820 diff --git a/papers/items/2026-2606-24839-grading-the-grader-lessons-from-evaluating-an-agentic-data-analysis-system.md b/papers/items/2026-2606-24839-grading-the-grader-lessons-from-evaluating-an-agentic-data-analysis-system.md new file mode 100644 index 0000000..53dd4d1 --- /dev/null +++ b/papers/items/2026-2606-24839-grading-the-grader-lessons-from-evaluating-an-agentic-data-analysis-system.md @@ -0,0 +1,62 @@ +# Paper: Grading the Grader: Lessons from Evaluating an Agentic Data Analysis System + +--- +type: paper +title: "Grading the Grader: Lessons from Evaluating an Agentic Data Analysis System" +authors: Tian Zheng, Kai-Tai Hsu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24839 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - stat.AP +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, computer-use, multi-agent, tool-use +- arXiv categories: cs.AI, stat.AP +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24839 diff --git a/papers/items/2026-2606-24855-openthoughts-agent-data-recipes-for-agentic-models.md b/papers/items/2026-2606-24855-openthoughts-agent-data-recipes-for-agentic-models.md new file mode 100644 index 0000000..d5d4a23 --- /dev/null +++ b/papers/items/2026-2606-24855-openthoughts-agent-data-recipes-for-agentic-models.md @@ -0,0 +1,59 @@ +# Paper: OpenThoughts-Agent: Data Recipes for Agentic Models + +--- +type: paper +title: "OpenThoughts-Agent: Data Recipes for Agentic Models" +authors: Negin Raoof, Richard Zhuang, Marianna Nezhurina, Etash Guha, Atula Tejaswi, Ryan Marten, Charlie F. Ruan, Tyler Griggs, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24855 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, rag +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24855 diff --git a/papers/items/2026-2606-24937-the-hitchhiker-s-guide-to-agentic-ai-from-foundations-to-systems.md b/papers/items/2026-2606-24937-the-hitchhiker-s-guide-to-agentic-ai-from-foundations-to-systems.md new file mode 100644 index 0000000..f9a8c71 --- /dev/null +++ b/papers/items/2026-2606-24937-the-hitchhiker-s-guide-to-agentic-ai-from-foundations-to-systems.md @@ -0,0 +1,68 @@ +# Paper: The Hitchhiker's Guide to Agentic AI: From Foundations to Systems + +--- +type: paper +title: "The Hitchhiker's Guide to Agentic AI: From Foundations to Systems" +authors: Haggai Roitman +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24937 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-22 +updated_at: 2026-06-22 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.IR + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 25 +collection_queries: agentic-ai, multi-agent-llm, rag-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, multi-agent-llm, rag-agent, tool-use +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL, cs.IR, cs.LG +- collection score: 25 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24937 diff --git a/papers/items/2026-2606-24976-diagnosing-and-mitigating-compounding-failures-in-agentic-persuasion-via-taxonom.md b/papers/items/2026-2606-24976-diagnosing-and-mitigating-compounding-failures-in-agentic-persuasion-via-taxonom.md new file mode 100644 index 0000000..76ca7f5 --- /dev/null +++ b/papers/items/2026-2606-24976-diagnosing-and-mitigating-compounding-failures-in-agentic-persuasion-via-taxonom.md @@ -0,0 +1,63 @@ +# Paper: Diagnosing and Mitigating Compounding Failures in Agentic Persuasion via Taxonomic Strategy Retrieval + +--- +type: paper +title: Diagnosing and Mitigating Compounding Failures in Agentic Persuasion via Taxonomic Strategy Retrieval +authors: Sana Ayromlou, Purvi Sehgal, Pradyumna Narayana +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.24976 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, multi-agent, planning, rag +- arXiv categories: cs.AI, cs.CL, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.24976 diff --git a/papers/items/2026-2606-25115-forget-to-improve-on-device-llm-agent-continual-learning-via-budget-curated-memo.md b/papers/items/2026-2606-25115-forget-to-improve-on-device-llm-agent-continual-learning-via-budget-curated-memo.md new file mode 100644 index 0000000..542aa1a --- /dev/null +++ b/papers/items/2026-2606-25115-forget-to-improve-on-device-llm-agent-continual-learning-via-budget-curated-memo.md @@ -0,0 +1,61 @@ +# Paper: Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory + +--- +type: paper +title: "Forget to Improve: On-Device LLM-Agent Continual Learning via Budget-Curated Memory" +authors: Beining Wu, Zihao Ding, Jun Huang, Yanxiao Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25115 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.NI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, embodied-agent, memory +- arXiv categories: cs.LG, cs.NI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25115 diff --git a/papers/items/2026-2606-25139-buildrix-an-open-platform-for-sharing-and-benchmarking-agentic-ai-skills-in-buil.md b/papers/items/2026-2606-25139-buildrix-an-open-platform-for-sharing-and-benchmarking-agentic-ai-skills-in-buil.md new file mode 100644 index 0000000..5a55b6d --- /dev/null +++ b/papers/items/2026-2606-25139-buildrix-an-open-platform-for-sharing-and-benchmarking-agentic-ai-skills-in-buil.md @@ -0,0 +1,60 @@ +# Paper: Buildrix: An Open Platform for Sharing and Benchmarking Agentic AI Skills in Building Engineering + +--- +type: paper +title: "Buildrix: An Open Platform for Sharing and Benchmarking Agentic AI Skills in Building Engineering" +authors: Zixin Jiang, Bing Dong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25139 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, tool-use, workflow-agent +- arXiv categories: eess.SY +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25139 diff --git a/papers/items/2026-2606-25161-trustmem-learning-trustworthy-memory-consolidation-for-llm-agents-with-long-term.md b/papers/items/2026-2606-25161-trustmem-learning-trustworthy-memory-consolidation-for-llm-agents-with-long-term.md new file mode 100644 index 0000000..ac901b8 --- /dev/null +++ b/papers/items/2026-2606-25161-trustmem-learning-trustworthy-memory-consolidation-for-llm-agents-with-long-term.md @@ -0,0 +1,63 @@ +# Paper: TRUSTMEM: Learning Trustworthy Memory Consolidation for LLM Agents with Long-Term Memory + +--- +type: paper +title: "TRUSTMEM: Learning Trustworthy Memory Consolidation for LLM Agents with Long-Term Memory" +authors: Tianyu Yang, Sudipta Paul, Vijay Srinivasan, Vivek Kulkarni, Srinivas Chappidi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25161 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, computer-use, memory, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25161 diff --git a/papers/items/2026-2606-25189-actplane-programmable-os-level-policy-enforcement-for-agent-harnesses.md b/papers/items/2026-2606-25189-actplane-programmable-os-level-policy-enforcement-for-agent-harnesses.md new file mode 100644 index 0000000..e4a1bcf --- /dev/null +++ b/papers/items/2026-2606-25189-actplane-programmable-os-level-policy-enforcement-for-agent-harnesses.md @@ -0,0 +1,62 @@ +# Paper: ActPlane: Programmable OS-Level Policy Enforcement for Agent Harnesses + +--- +type: paper +title: "ActPlane: Programmable OS-Level Policy Enforcement for Agent Harnesses" +authors: Yusheng Zheng, Tianyuan Wu, Quanzhi Fu, Tong Yu, Wenan Mao, Tao Ma, Dan Williams, Wei Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25189 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.OS +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, planning, tool-use +- arXiv categories: cs.OS +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25189 diff --git a/papers/items/2026-2606-25191-to-isolate-or-to-score-model-adaptive-assessment-for-cost-efficient-multi-agent-.md b/papers/items/2026-2606-25191-to-isolate-or-to-score-model-adaptive-assessment-for-cost-efficient-multi-agent-.md new file mode 100644 index 0000000..5524a93 --- /dev/null +++ b/papers/items/2026-2606-25191-to-isolate-or-to-score-model-adaptive-assessment-for-cost-efficient-multi-agent-.md @@ -0,0 +1,62 @@ +# Paper: To Isolate or to Score? Model-Adaptive Assessment for Cost-Efficient Multi-Agent RAG + +--- +type: paper +title: To Isolate or to Score? Model-Adaptive Assessment for Cost-Efficient Multi-Agent RAG +authors: Jungseob Lee, Chanjun Park, Heuiseok Lim +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25191 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, multi-agent, rag, reasoning +- arXiv categories: cs.AI, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25191 diff --git a/papers/items/2026-2606-25195-sok-ai-secure-code-generation-progress-pitfalls-and-paths-forward.md b/papers/items/2026-2606-25195-sok-ai-secure-code-generation-progress-pitfalls-and-paths-forward.md new file mode 100644 index 0000000..a28e5fb --- /dev/null +++ b/papers/items/2026-2606-25195-sok-ai-secure-code-generation-progress-pitfalls-and-paths-forward.md @@ -0,0 +1,63 @@ +# Paper: SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward + +--- +type: paper +title: "SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward" +authors: Rupam Patir, Keyan Guo, Haipeng Cai, Hongxin Hu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25195 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agentic-ai, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25195 diff --git a/papers/items/2026-2606-25206-raven-long-horizon-reasoning-navigation-with-a-visuo-spatio-temporal-memory.md b/papers/items/2026-2606-25206-raven-long-horizon-reasoning-navigation-with-a-visuo-spatio-temporal-memory.md new file mode 100644 index 0000000..01f441b --- /dev/null +++ b/papers/items/2026-2606-25206-raven-long-horizon-reasoning-navigation-with-a-visuo-spatio-temporal-memory.md @@ -0,0 +1,65 @@ +# Paper: RAVEN: Long-Horizon Reasoning & Navigation with a Visuo-Spatio-Temporal Memory + +--- +type: paper +title: "RAVEN: Long-Horizon Reasoning & Navigation with a Visuo-Spatio-Temporal Memory" +authors: Yixun Hu, Zhicheng Zheng, Lihan Zha, Chunwei Xing, Rajdeep Singh, Omar Hossain, Antonio Loquercio, Dhruv Shah +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25206 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-23 +updated_at: 2026-06-23 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, embodied-agent, memory, planning, rag, reasoning +- arXiv categories: cs.RO, cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25206 diff --git a/papers/items/2026-2606-25334-bridging-the-post-discharge-gap-a-traceable-multi-agent-framework-for-safe-and-c.md b/papers/items/2026-2606-25334-bridging-the-post-discharge-gap-a-traceable-multi-agent-framework-for-safe-and-c.md new file mode 100644 index 0000000..b21dffb --- /dev/null +++ b/papers/items/2026-2606-25334-bridging-the-post-discharge-gap-a-traceable-multi-agent-framework-for-safe-and-c.md @@ -0,0 +1,62 @@ +# Paper: Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care + +--- +type: paper +title: "Bridging the Post-discharge Gap: A Traceable Multi-agent Framework for Safe and Continuous Care" +authors: Runwei Guan, Yi Zhou, Heyi Lin, Jinjing Zhu, Mingyuan Hou, Yang Yang, Fang Yuan, Xiaohong Lin, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25334 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - multi-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, memory, multi-agent, rag +- arXiv categories: cs.MA +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25334 diff --git a/papers/items/2026-2606-25358-agentic-knowledge-tracing-a-multi-agent-llm-architecture-for-stealth-assessment-.md b/papers/items/2026-2606-25358-agentic-knowledge-tracing-a-multi-agent-llm-architecture-for-stealth-assessment-.md new file mode 100644 index 0000000..fb50a54 --- /dev/null +++ b/papers/items/2026-2606-25358-agentic-knowledge-tracing-a-multi-agent-llm-architecture-for-stealth-assessment-.md @@ -0,0 +1,63 @@ +# Paper: Agentic Knowledge Tracing: A Multi-Agent LLM Architecture for Stealth Assessment of Financial Literacy in Serious Games + +--- +type: paper +title: "Agentic Knowledge Tracing: A Multi-Agent LLM Architecture for Stealth Assessment of Financial Literacy in Serious Games" +authors: Gabriel Santos, Rita Julia, Marcelo Nascimento +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25358 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI, cs.MA +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25358 diff --git a/papers/items/2026-2606-25361-memory-makes-the-difference-evaluating-how-different-memory-roles-shape-conversa.md b/papers/items/2026-2606-25361-memory-makes-the-difference-evaluating-how-different-memory-roles-shape-conversa.md new file mode 100644 index 0000000..d934692 --- /dev/null +++ b/papers/items/2026-2606-25361-memory-makes-the-difference-evaluating-how-different-memory-roles-shape-conversa.md @@ -0,0 +1,63 @@ +# Paper: Memory Makes the Difference: Evaluating How Different Memory Roles Shape Conversational Agents + +--- +type: paper +title: "Memory Makes the Difference: Evaluating How Different Memory Roles Shape Conversational Agents" +authors: Yuxin Wang, Paul Thomas, Zhiwei Yu, Yuan Gao, Saeed Hassanpour, Soroush Vosoughi, Robert Sim, Nick Craswell +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25361 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.CL, cs.AI, cs.IR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25361 diff --git a/papers/items/2026-2606-25400-brainagent-a-large-language-model-driven-multi-agent-framework-for-autonomous-br.md b/papers/items/2026-2606-25400-brainagent-a-large-language-model-driven-multi-agent-framework-for-autonomous-br.md new file mode 100644 index 0000000..d8e29c7 --- /dev/null +++ b/papers/items/2026-2606-25400-brainagent-a-large-language-model-driven-multi-agent-framework-for-autonomous-br.md @@ -0,0 +1,62 @@ +# Paper: BrainAgent: A Large Language Model-Driven Multi-Agent Framework for Autonomous Brain Signal Understanding + +--- +type: paper +title: "BrainAgent: A Large Language Model-Driven Multi-Agent Framework for Autonomous Brain Signal Understanding" +authors: Yangxuan Zhou, Sha Zhao, Jiquan Wang, Shijian Li, Gang Pan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25400 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25400 diff --git a/papers/items/2026-2606-25484-from-causal-discovery-to-implementation-an-agentic-ai-framework-for-e-scooter-mo.md b/papers/items/2026-2606-25484-from-causal-discovery-to-implementation-an-agentic-ai-framework-for-e-scooter-mo.md new file mode 100644 index 0000000..b36d9bf --- /dev/null +++ b/papers/items/2026-2606-25484-from-causal-discovery-to-implementation-an-agentic-ai-framework-for-e-scooter-mo.md @@ -0,0 +1,63 @@ +# Paper: From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities + +--- +type: paper +title: "From Causal Discovery to Implementation: An Agentic AI Framework for E-Scooter Mobility Hub Planning Across 29 German Cities" +authors: Meng Jin, Melanie Handrich, Simone Martinenz, Nicholas Hoeser, Ziyue Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25484 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - coding-agent + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY + - econ.GN + - stat.AP +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: coding-agent, computer-use, planning, tool-use +- arXiv categories: cs.CY, econ.GN, stat.AP +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25484 diff --git a/papers/items/2026-2606-25514-unlocking-model-potentials-through-adaptive-multi-agent-scaffolding-for-efficien.md b/papers/items/2026-2606-25514-unlocking-model-potentials-through-adaptive-multi-agent-scaffolding-for-efficien.md new file mode 100644 index 0000000..2165ef4 --- /dev/null +++ b/papers/items/2026-2606-25514-unlocking-model-potentials-through-adaptive-multi-agent-scaffolding-for-efficien.md @@ -0,0 +1,64 @@ +# Paper: Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution + +--- +type: paper +title: Unlocking Model Potentials Through Adaptive Multi-Agent Scaffolding for Efficient Issue Resolution +authors: Yang Chen, Aliya Ahmad, Yiheng Zhou, Reyhaneh Jabbarvand +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25514 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, multi-agent, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25514 diff --git a/papers/items/2026-2606-25519-quantization-inflates-reasoning-token-inflation-as-a-hidden-cost-of-low-bit-reas.md b/papers/items/2026-2606-25519-quantization-inflates-reasoning-token-inflation-as-a-hidden-cost-of-low-bit-reas.md new file mode 100644 index 0000000..d9a21ff --- /dev/null +++ b/papers/items/2026-2606-25519-quantization-inflates-reasoning-token-inflation-as-a-hidden-cost-of-low-bit-reas.md @@ -0,0 +1,63 @@ +# Paper: Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models + +--- +type: paper +title: "Quantization Inflates Reasoning: Token Inflation as a Hidden Cost of Low-Bit Reasoning Models" +authors: Xinyu Lian, Walid Krichene, Beichen Huang, Masahiro Tanaka, Olatunji Ruwase, Li Zhang, Minjia Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25519 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, coding-agent, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25519 diff --git a/papers/items/2026-2606-25588-intenttester-intent-driven-multi-agent-framework-for-cross-library-test-migratio.md b/papers/items/2026-2606-25588-intenttester-intent-driven-multi-agent-framework-for-cross-library-test-migratio.md new file mode 100644 index 0000000..ee99132 --- /dev/null +++ b/papers/items/2026-2606-25588-intenttester-intent-driven-multi-agent-framework-for-cross-library-test-migratio.md @@ -0,0 +1,64 @@ +# Paper: IntentTester: Intent-Driven Multi-agent Framework for Cross-Library Test Migration + +--- +type: paper +title: "IntentTester: Intent-Driven Multi-agent Framework for Cross-Library Test Migration" +authors: Yi Gao, Ziyuan Zhang, Xing Hu, Xiaohu Yang, Xin Xia +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25588 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, multi-agent, reasoning, tool-use +- arXiv categories: cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25588 diff --git a/papers/items/2026-2606-25622-probabilistic-agents-in-deterministic-audits-evaluating-multi-agent-systems-for-.md b/papers/items/2026-2606-25622-probabilistic-agents-in-deterministic-audits-evaluating-multi-agent-systems-for-.md new file mode 100644 index 0000000..0f8f9f0 --- /dev/null +++ b/papers/items/2026-2606-25622-probabilistic-agents-in-deterministic-audits-evaluating-multi-agent-systems-for-.md @@ -0,0 +1,65 @@ +# Paper: Probabilistic Agents in Deterministic Audits: Evaluating Multi-Agent Systems for Automated Audits Based on the German IT-Grundschutz + +--- +type: paper +title: "Probabilistic Agents in Deterministic Audits: Evaluating Multi-Agent Systems for Automated Audits Based on the German IT-Grundschutz" +authors: Lea Roxanne Muth, Marian Margraf +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25622 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25622 diff --git a/papers/items/2026-2606-25651-medguards-multi-agent-system-for-reliable-medical-error-detection-and-correction.md b/papers/items/2026-2606-25651-medguards-multi-agent-system-for-reliable-medical-error-detection-and-correction.md new file mode 100644 index 0000000..5f6b564 --- /dev/null +++ b/papers/items/2026-2606-25651-medguards-multi-agent-system-for-reliable-medical-error-detection-and-correction.md @@ -0,0 +1,62 @@ +# Paper: MedGuards: Multi-Agent System for Reliable Medical Error Detection and Correction + +--- +type: paper +title: "MedGuards: Multi-Agent System for Reliable Medical Error Detection and Correction" +authors: Congbo Ma, Hu Wang, Yichun Zhang, Farah E. Shamout +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25651 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, reasoning +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25651 diff --git a/papers/items/2026-2606-25656-is-graphrag-needed-from-basic-rag-to-graph-agentic-solutions-with-context-optimi.md b/papers/items/2026-2606-25656-is-graphrag-needed-from-basic-rag-to-graph-agentic-solutions-with-context-optimi.md new file mode 100644 index 0000000..06150fa --- /dev/null +++ b/papers/items/2026-2606-25656-is-graphrag-needed-from-basic-rag-to-graph-agentic-solutions-with-context-optimi.md @@ -0,0 +1,63 @@ +# Paper: Is GraphRAG Needed? From Basic RAG to Graph-/Agentic Solutions with Context Optimization + +--- +type: paper +title: Is GraphRAG Needed? From Basic RAG to Graph-/Agentic Solutions with Context Optimization +authors: Long Chen, Ryan Razkenari, Yuxuan Zhou, Yuan Tian, Rahul Ghosh, Venkatesh Pappakrishnan, Disha Ahuja, Vidya Sagar Ravipati +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25656 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, memory, planning, rag +- arXiv categories: cs.CL, cs.AI, cs.IR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25656 diff --git a/papers/items/2026-2606-25705-gui-agent-guided-exploration-of-user-sensitive-screens.md b/papers/items/2026-2606-25705-gui-agent-guided-exploration-of-user-sensitive-screens.md new file mode 100644 index 0000000..591f31c --- /dev/null +++ b/papers/items/2026-2606-25705-gui-agent-guided-exploration-of-user-sensitive-screens.md @@ -0,0 +1,60 @@ +# Paper: GUI agent: Guided Exploration of User-Sensitive Screens + +--- +type: paper +title: "GUI agent: Guided Exploration of User-Sensitive Screens" +authors: Aradhana Nayak, Mussadiq Nazeer, Wang Peng, Feng Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25705 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-safety, computer-use, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25705 diff --git a/papers/items/2026-2606-25760-uncertainty-quantification-for-computer-use-agents-a-benchmark-across-vision-lan.md b/papers/items/2026-2606-25760-uncertainty-quantification-for-computer-use-agents-a-benchmark-across-vision-lan.md new file mode 100644 index 0000000..0579093 --- /dev/null +++ b/papers/items/2026-2606-25760-uncertainty-quantification-for-computer-use-agents-a-benchmark-across-vision-lan.md @@ -0,0 +1,64 @@ +# Paper: Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets + +--- +type: paper +title: "Uncertainty Quantification for Computer-Use Agents: A Benchmark across Vision-Language Models and GUI Grounding Datasets" +authors: Divake Kumar, Sina Tayebati, Devashri Naik, Amanda Sofie Rios, Nilesh Ahuja, Omesh Tickoo, Ranganath Krishnan, Amit Ranjan Trivedi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25760 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, web-gui-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, rag +- arXiv categories: cs.LG, cs.AI, cs.CL, cs.CV +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25760 diff --git a/papers/items/2026-2606-25819-beyond-function-calling-benchmarking-tool-using-agents-under-tool-environment-un.md b/papers/items/2026-2606-25819-beyond-function-calling-benchmarking-tool-using-agents-under-tool-environment-un.md new file mode 100644 index 0000000..8dfd06a --- /dev/null +++ b/papers/items/2026-2606-25819-beyond-function-calling-benchmarking-tool-using-agents-under-tool-environment-un.md @@ -0,0 +1,61 @@ +# Paper: Beyond Function Calling: Benchmarking Tool-Using Agents under Tool-Environment Unreliability + +--- +type: paper +title: "Beyond Function Calling: Benchmarking Tool-Using Agents under Tool-Environment Unreliability" +authors: Yang Tian, Zhengpeng Shi, Yu Zhou, Bo Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25819 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling, tool-use +- inferred topics: agent-evaluation, tool-use, workflow-agent +- arXiv categories: cs.CL, cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25819 diff --git a/papers/items/2026-2606-25899-manipulation-is-task-dependent-a-multi-axis-multi-environment-evaluation-of-fron.md b/papers/items/2026-2606-25899-manipulation-is-task-dependent-a-multi-axis-multi-environment-evaluation-of-fron.md new file mode 100644 index 0000000..84877ae --- /dev/null +++ b/papers/items/2026-2606-25899-manipulation-is-task-dependent-a-multi-axis-multi-environment-evaluation-of-fron.md @@ -0,0 +1,61 @@ +# Paper: Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs + +--- +type: paper +title: "Manipulation Is Task-Dependent: A Multi-Axis, Multi-Environment Evaluation of Frontier LLMs" +authors: Adeeb Zaman, Erik Nordby, Fred Heiding +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.25899 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, rag, tool-use, workflow-agent +- arXiv categories: cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.25899 diff --git a/papers/items/2026-2606-26057-the-unfireable-safety-kernel-execution-time-ai-alignment-for-ai-agents-and-other.md b/papers/items/2026-2606-26057-the-unfireable-safety-kernel-execution-time-ai-alignment-for-ai-agents-and-other.md new file mode 100644 index 0000000..db11a3c --- /dev/null +++ b/papers/items/2026-2606-26057-the-unfireable-safety-kernel-execution-time-ai-alignment-for-ai-agents-and-other.md @@ -0,0 +1,64 @@ +# Paper: The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems + +--- +type: paper +title: "The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems" +authors: Seth Dobrin, Łukasz Chmiel +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26057 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CR + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, planning, tool-use, world-model +- arXiv categories: cs.AI, cs.CR, cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26057 diff --git a/papers/items/2026-2606-26203-agentic-analysis-for-agentic-infrastructure-an-llm-powered-pipeline-for-comparat.md b/papers/items/2026-2606-26203-agentic-analysis-for-agentic-infrastructure-an-llm-powered-pipeline-for-comparat.md new file mode 100644 index 0000000..63bb12f --- /dev/null +++ b/papers/items/2026-2606-26203-agentic-analysis-for-agentic-infrastructure-an-llm-powered-pipeline-for-comparat.md @@ -0,0 +1,64 @@ +# Paper: Agentic Analysis for Agentic Infrastructure: An LLM-Powered Pipeline for Comparative Governance of DAO and Corporate AI Protocols + +--- +type: paper +title: "Agentic Analysis for Agentic Infrastructure: An LLM-Powered Pipeline for Comparative Governance of DAO and Corporate AI Protocols" +authors: Yutian Wang, Luyao Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26203 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CY + - cs.MA + - cs.SI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai, ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, ai-agent +- inferred topics: agent-safety, coding-agent, rag, tool-use +- arXiv categories: cs.AI, cs.CY, cs.MA, cs.SI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26203 diff --git a/papers/items/2026-2606-26205-knowledge-augmented-agentic-ai-for-mental-health-medication-information-seeking.md b/papers/items/2026-2606-26205-knowledge-augmented-agentic-ai-for-mental-health-medication-information-seeking.md new file mode 100644 index 0000000..3cd9535 --- /dev/null +++ b/papers/items/2026-2606-26205-knowledge-augmented-agentic-ai-for-mental-health-medication-information-seeking.md @@ -0,0 +1,60 @@ +# Paper: Knowledge-augmented Agentic AI for Mental Health Medication Information Seeking + +--- +type: paper +title: Knowledge-augmented Agentic AI for Mental Health Medication Information Seeking +authors: Huizi Yu, Jian Liu, Wenkong Wang, Lingyao Li, Jiayan Zhou, Zhaoqian Xue, Xiang Li, Xinxin Lin, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26205 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26205 diff --git a/papers/items/2026-2606-26216-cyberchainbench-can-ai-agents-secure-smart-contracts-against-real-world-on-chain.md b/papers/items/2026-2606-26216-cyberchainbench-can-ai-agents-secure-smart-contracts-against-real-world-on-chain.md new file mode 100644 index 0000000..c4d2978 --- /dev/null +++ b/papers/items/2026-2606-26216-cyberchainbench-can-ai-agents-secure-smart-contracts-against-real-world-on-chain.md @@ -0,0 +1,61 @@ +# Paper: CyberChainBench: Can AI Agents Secure Smart Contracts Against Real-World On-Chain Vulnerabilities? + +--- +type: paper +title: "CyberChainBench: Can AI Agents Secure Smart Contracts Against Real-World On-Chain Vulnerabilities?" +authors: Jintao Huang, Fengqing Jiang, Radha Poovendran, Zhiqiang Lin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26216 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety, ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, ai-agent +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26216 diff --git a/papers/items/2026-2606-26300-the-verification-horizon-no-silver-bullet-for-coding-agent-rewards.md b/papers/items/2026-2606-26300-the-verification-horizon-no-silver-bullet-for-coding-agent-rewards.md new file mode 100644 index 0000000..d858dc8 --- /dev/null +++ b/papers/items/2026-2606-26300-the-verification-horizon-no-silver-bullet-for-coding-agent-rewards.md @@ -0,0 +1,63 @@ +# Paper: The Verification Horizon: No Silver Bullet for Coding Agent Rewards + +--- +type: paper +title: "The Verification Horizon: No Silver Bullet for Coding Agent Rewards" +authors: Binghai Wang, Chenlong Zhang, Dayiheng Liu, Jiajun Zhang, Jiawei Chen, Mingze Li, Mouxiang Chen, Rongyao Fang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26300 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, planning, rag, reasoning +- arXiv categories: cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26300 diff --git a/papers/items/2026-2606-26346-how-do-tool-augmented-llm-agents-perform-on-real-world-energy-analytics-tasks.md b/papers/items/2026-2606-26346-how-do-tool-augmented-llm-agents-perform-on-real-world-energy-analytics-tasks.md new file mode 100644 index 0000000..ed27959 --- /dev/null +++ b/papers/items/2026-2606-26346-how-do-tool-augmented-llm-agents-perform-on-real-world-energy-analytics-tasks.md @@ -0,0 +1,63 @@ +# Paper: How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks? + +--- +type: paper +title: How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks? +authors: David Akinpelu, Akintonde Abbas, Rereloluwa Alimi, Ayodeji Lana +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26346 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, coding-agent, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26346 diff --git a/papers/items/2026-2606-26356-instruction-bleed-cross-module-interference-in-prompt-composed-agentic-systems.md b/papers/items/2026-2606-26356-instruction-bleed-cross-module-interference-in-prompt-composed-agentic-systems.md new file mode 100644 index 0000000..1242ccb --- /dev/null +++ b/papers/items/2026-2606-26356-instruction-bleed-cross-module-interference-in-prompt-composed-agentic-systems.md @@ -0,0 +1,61 @@ +# Paper: Instruction Bleed: Cross-Module Interference in Prompt-Composed Agentic Systems + +--- +type: paper +title: "Instruction Bleed: Cross-Module Interference in Prompt-Composed Agentic Systems" +authors: Ching-Yu Lin, Yifan Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26356 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.IR + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation, agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, agentic-ai +- inferred topics: agent-evaluation, multi-agent +- arXiv categories: cs.AI, cs.IR, cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26356 diff --git a/papers/items/2026-2606-26403-profilefoundry-a-synthetic-person-object-substrate-for-privacy-memory-and-tool-u.md b/papers/items/2026-2606-26403-profilefoundry-a-synthetic-person-object-substrate-for-privacy-memory-and-tool-u.md new file mode 100644 index 0000000..557cab3 --- /dev/null +++ b/papers/items/2026-2606-26403-profilefoundry-a-synthetic-person-object-substrate-for-privacy-memory-and-tool-u.md @@ -0,0 +1,60 @@ +# Paper: ProfileFoundry: A Synthetic Person-Object Substrate for Privacy, Memory, and Tool-Use Evaluation in LLM Agent + +--- +type: paper +title: "ProfileFoundry: A Synthetic Person-Object Substrate for Privacy, Memory, and Tool-Use Evaluation in LLM Agent" +authors: Sriram Selvam, Anneswa Ghosh +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26403 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, memory, tool-use +- arXiv categories: cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26403 diff --git a/papers/items/2026-2606-26453-optimizing-cuda-like-a-human-micro-profiling-tools-as-expert-surrogates-for-llm-.md b/papers/items/2026-2606-26453-optimizing-cuda-like-a-human-micro-profiling-tools-as-expert-surrogates-for-llm-.md new file mode 100644 index 0000000..67a8264 --- /dev/null +++ b/papers/items/2026-2606-26453-optimizing-cuda-like-a-human-micro-profiling-tools-as-expert-surrogates-for-llm-.md @@ -0,0 +1,63 @@ +# Paper: Optimizing CUDA like a Human: Micro-Profiling Tools as Expert Surrogates for LLM-Based GPU Kernel Optimization + +--- +type: paper +title: "Optimizing CUDA like a Human: Micro-Profiling Tools as Expert Surrogates for LLM-Based GPU Kernel Optimization" +authors: Jiading Gai, Shuai Zhang, Kaj Bostrom, Jin Huang, Vihang Patil, Haoyang Fang, Bernie Wang, Huzefa Rangwala, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26453 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - memory + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, computer-use, memory, multi-agent, tool-use +- arXiv categories: cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26453 diff --git a/papers/items/2026-2606-26479-adaptive-evaluation-of-out-of-band-defenses-against-prompt-injection-in-llm-agen.md b/papers/items/2026-2606-26479-adaptive-evaluation-of-out-of-band-defenses-against-prompt-injection-in-llm-agen.md new file mode 100644 index 0000000..2effe82 --- /dev/null +++ b/papers/items/2026-2606-26479-adaptive-evaluation-of-out-of-band-defenses-against-prompt-injection-in-llm-agen.md @@ -0,0 +1,64 @@ +# Paper: Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents + +--- +type: paper +title: Adaptive Evaluation of Out-of-Band Defenses Against Prompt Injection in LLM Agents +authors: Praneeth Narisetty, Shiva Nagendra Babu Kore, Uday Kumar Reddy Kattamanchi, Jayaram Kumarapu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26479 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.CR, cs.AI, cs.CL, cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26479 diff --git a/papers/items/2026-2606-26511-temporal-validity-in-retrieval-memory-eliminating-stale-fact-errors-for-ai-agent.md b/papers/items/2026-2606-26511-temporal-validity-in-retrieval-memory-eliminating-stale-fact-errors-for-ai-agent.md new file mode 100644 index 0000000..4e69458 --- /dev/null +++ b/papers/items/2026-2606-26511-temporal-validity-in-retrieval-memory-eliminating-stale-fact-errors-for-ai-agent.md @@ -0,0 +1,65 @@ +# Paper: Temporal Validity in Retrieval Memory: Eliminating Stale-Fact Errors for AI Agents over Evolving Knowledge + +--- +type: paper +title: "Temporal Validity in Retrieval Memory: Eliminating Stale-Fact Errors for AI Agents over Evolving Knowledge" +authors: Neeraj Yadav +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26511 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.ET + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: ai-agent, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, rag-agent +- inferred topics: agent-evaluation, computer-use, memory, rag, tool-use +- arXiv categories: cs.CL, cs.AI, cs.ET, cs.LG +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26511 diff --git a/papers/items/2026-2606-26524-vigil-runtime-enforcement-of-behavioral-specifications-in-ai-agent-skills.md b/papers/items/2026-2606-26524-vigil-runtime-enforcement-of-behavioral-specifications-in-ai-agent-skills.md new file mode 100644 index 0000000..3cb867c --- /dev/null +++ b/papers/items/2026-2606-26524-vigil-runtime-enforcement-of-behavioral-specifications-in-ai-agent-skills.md @@ -0,0 +1,60 @@ +# Paper: VIGIL: Runtime Enforcement of Behavioral Specifications in AI Agent Skills + +--- +type: paper +title: "VIGIL: Runtime Enforcement of Behavioral Specifications in AI Agent Skills" +authors: Ying Li, Yanju Chen, Hongbo Wen, Bosi Zhang, Hanzhi Liu, Peiran Wang, Yu Feng, Yuan Tian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26524 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, tool-use, workflow-agent +- arXiv categories: cs.CR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26524 diff --git a/papers/items/2026-2606-26614-hilsva-design-and-evaluation-of-a-human-in-the-loop-agentic-system-for-scientifi.md b/papers/items/2026-2606-26614-hilsva-design-and-evaluation-of-a-human-in-the-loop-agentic-system-for-scientifi.md new file mode 100644 index 0000000..5038903 --- /dev/null +++ b/papers/items/2026-2606-26614-hilsva-design-and-evaluation-of-a-human-in-the-loop-agentic-system-for-scientifi.md @@ -0,0 +1,67 @@ +# Paper: HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization + +--- +type: paper +title: "HiLSVA: Design and Evaluation of a Human-in-the-Loop Agentic System for Scientific Visualization" +authors: Kuangshi Ai, Patrick Phuoc Do, Chaoli Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26614 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.HC + - cs.AI + - cs.GR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 23 +collection_queries: multi-agent-llm, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm, planning-agent +- inferred topics: agent-evaluation, computer-use, multi-agent, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.HC, cs.AI, cs.GR +- collection score: 23 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26614 diff --git a/papers/items/2026-2606-26627-agents-that-know-too-much-a-data-centric-survey-of-privacy-in-llm-agents.md b/papers/items/2026-2606-26627-agents-that-know-too-much-a-data-centric-survey-of-privacy-in-llm-agents.md new file mode 100644 index 0000000..743f6ba --- /dev/null +++ b/papers/items/2026-2606-26627-agents-that-know-too-much-a-data-centric-survey-of-privacy-in-llm-agents.md @@ -0,0 +1,64 @@ +# Paper: Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents + +--- +type: paper +title: "Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents" +authors: Nada Lahjouji, Ashwin Gerard Colaco +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26627 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, memory, rag, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26627 diff --git a/papers/items/2026-2606-26649-autoformalization-of-agent-instructions-into-policy-as-code.md b/papers/items/2026-2606-26649-autoformalization-of-agent-instructions-into-policy-as-code.md new file mode 100644 index 0000000..eb9604d --- /dev/null +++ b/papers/items/2026-2606-26649-autoformalization-of-agent-instructions-into-policy-as-code.md @@ -0,0 +1,62 @@ +# Paper: Autoformalization of Agent Instructions into Policy-as-Code + +--- +type: paper +title: Autoformalization of Agent Instructions into Policy-as-Code +authors: Adam Mondl, Matthew Maisel, John H. Brock +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26649 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.AI, cs.CR +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26649 diff --git a/papers/items/2026-2606-26721-knowledge-based-pull-requests-a-trusted-workflow-for-agent-mediated-knowledge-co.md b/papers/items/2026-2606-26721-knowledge-based-pull-requests-a-trusted-workflow-for-agent-mediated-knowledge-co.md new file mode 100644 index 0000000..ee821ed --- /dev/null +++ b/papers/items/2026-2606-26721-knowledge-based-pull-requests-a-trusted-workflow-for-agent-mediated-knowledge-co.md @@ -0,0 +1,67 @@ +# Paper: Knowledge-Based Pull Requests: A Trusted Workflow for Agent-Mediated Knowledge Collaboration + +--- +type: paper +title: "Knowledge-Based Pull Requests: A Trusted Workflow for Agent-Mediated Knowledge Collaboration" +authors: Xinyu Zhang, Weiwei Sun +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26721 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - multi-agent + - planning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, multi-agent, planning, tool-use, workflow-agent, world-model +- arXiv categories: cs.SE, cs.HC +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26721 diff --git a/papers/items/2026-2606-26758-egg-an-expert-guided-agent-framework-for-kernel-generation.md b/papers/items/2026-2606-26758-egg-an-expert-guided-agent-framework-for-kernel-generation.md new file mode 100644 index 0000000..95e1588 --- /dev/null +++ b/papers/items/2026-2606-26758-egg-an-expert-guided-agent-framework-for-kernel-generation.md @@ -0,0 +1,62 @@ +# Paper: EGG: An Expert-Guided Agent Framework for Kernel Generation + +--- +type: paper +title: "EGG: An Expert-Guided Agent Framework for Kernel Generation" +authors: Yaochen Han, Ke Fan, Hongxu Jiang, Wanqi Xu, Weiyu Xie, Runhua Zhang, Chenhui Zhu, Yixiang Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26758 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - computer-use + - memory + - multi-agent + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: computer-use, memory, multi-agent, rag, workflow-agent +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26758 diff --git a/papers/items/2026-2606-26793-mirror-novelty-constrained-memory-guided-mcts-red-teaming-for-agentic-rag.md b/papers/items/2026-2606-26793-mirror-novelty-constrained-memory-guided-mcts-red-teaming-for-agentic-rag.md new file mode 100644 index 0000000..35dfdf8 --- /dev/null +++ b/papers/items/2026-2606-26793-mirror-novelty-constrained-memory-guided-mcts-red-teaming-for-agentic-rag.md @@ -0,0 +1,65 @@ +# Paper: MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG + +--- +type: paper +title: "MIRROR: Novelty-Constrained Memory-Guided MCTS Red-Teaming for Agentic RAG" +authors: Inderjeet Singh, Andrés Murillo, Motoyoshi Sekiya, Yuki Unno, Junichi Suga +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26793 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, rag, tool-use +- arXiv categories: cs.CR, cs.AI, cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26793 diff --git a/papers/items/2026-2606-26806-memory-depth-not-memory-access-selective-parametric-consolidation-for-long-runni.md b/papers/items/2026-2606-26806-memory-depth-not-memory-access-selective-parametric-consolidation-for-long-runni.md new file mode 100644 index 0000000..0f316fd --- /dev/null +++ b/papers/items/2026-2606-26806-memory-depth-not-memory-access-selective-parametric-consolidation-for-long-runni.md @@ -0,0 +1,61 @@ +# Paper: Memory Depth, Not Memory Access: Selective Parametric Consolidation for Long-Running Language Agents + +--- +type: paper +title: "Memory Depth, Not Memory Access: Selective Parametric Consolidation for Long-Running Language Agents" +authors: Haoliang Han +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26806 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, rag +- arXiv categories: cs.AI, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26806 diff --git a/papers/items/2026-2606-26883-econsimulacra-a-digital-twin-platform-of-socio-economic-systems-powered-by-llm-a.md b/papers/items/2026-2606-26883-econsimulacra-a-digital-twin-platform-of-socio-economic-systems-powered-by-llm-a.md new file mode 100644 index 0000000..3ca557a --- /dev/null +++ b/papers/items/2026-2606-26883-econsimulacra-a-digital-twin-platform-of-socio-economic-systems-powered-by-llm-a.md @@ -0,0 +1,60 @@ +# Paper: EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents + +--- +type: paper +title: "EconSimulacra: A Digital Twin Platform of Socio-Economic Systems Powered by LLM Agents" +authors: Ryuji Hashimoto, Masahiro Kaneko, Kentaro Ueda, Takehiro Takayanagi, Kiyoshi Izumi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26883 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - memory + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: memory, multi-agent, tool-use +- arXiv categories: cs.DL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26883 diff --git a/papers/items/2026-2606-26918-diagnosing-task-insensitivity-in-language-agents.md b/papers/items/2026-2606-26918-diagnosing-task-insensitivity-in-language-agents.md new file mode 100644 index 0000000..ea68977 --- /dev/null +++ b/papers/items/2026-2606-26918-diagnosing-task-insensitivity-in-language-agents.md @@ -0,0 +1,61 @@ +# Paper: Diagnosing Task Insensitivity in Language Agents + +--- +type: paper +title: Diagnosing Task Insensitivity in Language Agents +authors: Jingyu Liu, Xiaopeng Wu, Kehan Chen, Chuan Yu, Yong Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26918 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26918 diff --git a/papers/items/2026-2606-26924-a-deterministic-control-plane-for-llm-coding-agents.md b/papers/items/2026-2606-26924-a-deterministic-control-plane-for-llm-coding-agents.md new file mode 100644 index 0000000..546ce13 --- /dev/null +++ b/papers/items/2026-2606-26924-a-deterministic-control-plane-for-llm-coding-agents.md @@ -0,0 +1,64 @@ +# Paper: A Deterministic Control Plane for LLM Coding Agents + +--- +type: paper +title: A Deterministic Control Plane for LLM Coding Agents +authors: Padmaraj Madatha +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26924 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.SE, cs.AI, cs.CR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26924 diff --git a/papers/items/2026-2606-26960-toward-agentic-sysadmin-rethinking-system-administration-with-ai-agents.md b/papers/items/2026-2606-26960-toward-agentic-sysadmin-rethinking-system-administration-with-ai-agents.md new file mode 100644 index 0000000..b92a620 --- /dev/null +++ b/papers/items/2026-2606-26960-toward-agentic-sysadmin-rethinking-system-administration-with-ai-agents.md @@ -0,0 +1,61 @@ +# Paper: Toward Agentic SysAdmin: Rethinking System Administration with AI Agents + +--- +type: paper +title: "Toward Agentic SysAdmin: Rethinking System Administration with AI Agents" +authors: Gianmaria Frigo, Davide Saladino, Alberto Castagnaro, Francesco Marchiori, Denis Donadel, Luca Pajola, Mauro Conti +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.26960 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.NI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.NI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.26960 diff --git a/papers/items/2026-2606-27009-semantic-early-stopping-for-iterative-llm-agent-loops.md b/papers/items/2026-2606-27009-semantic-early-stopping-for-iterative-llm-agent-loops.md new file mode 100644 index 0000000..5103faf --- /dev/null +++ b/papers/items/2026-2606-27009-semantic-early-stopping-for-iterative-llm-agent-loops.md @@ -0,0 +1,63 @@ +# Paper: Semantic Early-Stopping for Iterative LLM Agent Loops + +--- +type: paper +title: Semantic Early-Stopping for Iterative LLM Agent Loops +authors: Sahil Shrivastava +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27009 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, rag, tool-use +- arXiv categories: cs.AI, cs.LG, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27009 diff --git a/papers/items/2026-2606-27154-openrca-2-0-from-outcome-labels-to-causal-process-supervision.md b/papers/items/2026-2606-27154-openrca-2-0-from-outcome-labels-to-causal-process-supervision.md new file mode 100644 index 0000000..fc7b768 --- /dev/null +++ b/papers/items/2026-2606-27154-openrca-2-0-from-outcome-labels-to-causal-process-supervision.md @@ -0,0 +1,61 @@ +# Paper: OpenRCA 2.0: From Outcome Labels to Causal Process Supervision + +--- +type: paper +title: "OpenRCA 2.0: From Outcome Labels to Causal Process Supervision" +authors: "Aoyang Fang, Yifan Yang, Jin'ao Shang, Qisheng Lu, Junjielung Xu, Rui Wang, Songhan Zhang, Yuzhong Zhang, et al." +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27154 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27154 diff --git a/papers/items/2026-2606-27243-nova-a-verification-aware-agent-harness-for-architecture-evolution-in-industrial.md b/papers/items/2026-2606-27243-nova-a-verification-aware-agent-harness-for-architecture-evolution-in-industrial.md new file mode 100644 index 0000000..55b2d47 --- /dev/null +++ b/papers/items/2026-2606-27243-nova-a-verification-aware-agent-harness-for-architecture-evolution-in-industrial.md @@ -0,0 +1,64 @@ +# Paper: NOVA: A Verification-Aware Agent Harness for Architecture Evolution in Industrial Recommender Systems + +--- +type: paper +title: "NOVA: A Verification-Aware Agent Harness for Architecture Evolution in Industrial Recommender Systems" +authors: Shaohua Liu, Liang Fang, Yilong Sun, Shudong Huang, Qingsong Luo, Shaoxin Liu, Xiaoyang Chen, Dongqiang Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27243 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - memory + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, memory, workflow-agent +- arXiv categories: cs.IR, cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27243 diff --git a/papers/items/2026-2606-27330-empowering-gui-agents-via-autonomous-experience-exploration-and-hindsight-experi.md b/papers/items/2026-2606-27330-empowering-gui-agents-via-autonomous-experience-exploration-and-hindsight-experi.md new file mode 100644 index 0000000..e769711 --- /dev/null +++ b/papers/items/2026-2606-27330-empowering-gui-agents-via-autonomous-experience-exploration-and-hindsight-experi.md @@ -0,0 +1,65 @@ +# Paper: Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning + +--- +type: paper +title: Empowering GUI Agents via Autonomous Experience Exploration and Hindsight Experience Utilization for Task Planning +authors: Tianyi Men, Zhuoran Jin, Pengfei Cao, Yubo Chen, Kang Liu, Jun Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27330 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.CV + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, rag, tool-use +- arXiv categories: cs.CL, cs.AI, cs.CV, cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27330 diff --git a/papers/items/2026-2606-27350-chia-an-open-source-framework-for-principled-agentic-ai-driven-hardware-software.md b/papers/items/2026-2606-27350-chia-an-open-source-framework-for-principled-agentic-ai-driven-hardware-software.md new file mode 100644 index 0000000..0670d43 --- /dev/null +++ b/papers/items/2026-2606-27350-chia-an-open-source-framework-for-principled-agentic-ai-driven-hardware-software.md @@ -0,0 +1,61 @@ +# Paper: CHIA: An open-source framework for principled, agentic AI-driven hardware/software co-design research + +--- +type: paper +title: "CHIA: An open-source framework for principled, agentic AI-driven hardware/software co-design research" +authors: Angela Cui, Ferran Hermida-Rivera, Jack Toubes, Raghav Gupta, Jim Fang, Chengyi Lux Zhang, Ella Schwarz, Junha Kim, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27350 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, coding-agent +- inferred topics: agent-safety, coding-agent, tool-use, workflow-agent +- arXiv categories: cs.AR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27350 diff --git a/papers/items/2026-2606-27397-sidconarena-an-environment-evaluating-agents-in-open-ended-positive-sum-bargaini.md b/papers/items/2026-2606-27397-sidconarena-an-environment-evaluating-agents-in-open-ended-positive-sum-bargaini.md new file mode 100644 index 0000000..3a33d4c --- /dev/null +++ b/papers/items/2026-2606-27397-sidconarena-an-environment-evaluating-agents-in-open-ended-positive-sum-bargaini.md @@ -0,0 +1,64 @@ +# Paper: SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game + +--- +type: paper +title: "SidConArena: An Environment Evaluating Agents in Open-Ended,Positive-Sum Bargaining Game" +authors: Yeqi Feng, Yuxin Chen, Tianxing He +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27397 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-24 +updated_at: 2026-06-24 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.AI + - cs.GT +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, memory, planning, reasoning, tool-use +- arXiv categories: cs.MA, cs.AI, cs.GT +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27397 diff --git a/papers/items/2026-2606-27406-towards-evaluation-of-implicit-software-world-models-in-coding-llms.md b/papers/items/2026-2606-27406-towards-evaluation-of-implicit-software-world-models-in-coding-llms.md new file mode 100644 index 0000000..206562b --- /dev/null +++ b/papers/items/2026-2606-27406-towards-evaluation-of-implicit-software-world-models-in-coding-llms.md @@ -0,0 +1,63 @@ +# Paper: Towards Evaluation of Implicit Software World Models in Coding LLMs + +--- +type: paper +title: Towards Evaluation of Implicit Software World Models in Coding LLMs +authors: Egor Bogomolov, Yaroslav Zharov +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27406 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - reasoning + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: ai-agent, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, coding-agent +- inferred topics: agent-evaluation, coding-agent, memory, reasoning, world-model +- arXiv categories: cs.SE, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27406 diff --git a/papers/items/2026-2606-27416-glite-arf-verifier-driven-research-with-parallel-llm-coding-agents.md b/papers/items/2026-2606-27416-glite-arf-verifier-driven-research-with-parallel-llm-coding-agents.md new file mode 100644 index 0000000..4a3d259 --- /dev/null +++ b/papers/items/2026-2606-27416-glite-arf-verifier-driven-research-with-parallel-llm-coding-agents.md @@ -0,0 +1,61 @@ +# Paper: Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents + +--- +type: paper +title: "Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents" +authors: Vassili Philippov, Pavel Katunin, Dmitry Andreev, Igor Ostanin, Anton Nikolaev +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27416 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - coding-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: coding-agent, reasoning, tool-use +- arXiv categories: cs.MA, cs.SE +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27416 diff --git a/papers/items/2026-2606-27472-supersede-diagnosing-and-training-the-memory-update-gap-in-llm-agents.md b/papers/items/2026-2606-27472-supersede-diagnosing-and-training-the-memory-update-gap-in-llm-agents.md new file mode 100644 index 0000000..973034c --- /dev/null +++ b/papers/items/2026-2606-27472-supersede-diagnosing-and-training-the-memory-update-gap-in-llm-agents.md @@ -0,0 +1,64 @@ +# Paper: Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents + +--- +type: paper +title: "Supersede: Diagnosing and Training the Memory-Update Gap in LLM Agents" +authors: Vedant Patel +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27472 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, memory, planning, reasoning, tool-use +- arXiv categories: cs.CL, cs.AI, cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27472 diff --git a/papers/items/2026-2606-27483-internalizing-the-future-a-unified-agentic-training-paradigm-for-world-model-pla.md b/papers/items/2026-2606-27483-internalizing-the-future-a-unified-agentic-training-paradigm-for-world-model-pla.md new file mode 100644 index 0000000..36d2026 --- /dev/null +++ b/papers/items/2026-2606-27483-internalizing-the-future-a-unified-agentic-training-paradigm-for-world-model-pla.md @@ -0,0 +1,62 @@ +# Paper: Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning + +--- +type: paper +title: "Internalizing the Future: A Unified Agentic Training Paradigm for World Model Planning" +authors: Xuan Zhang, Zhijian Zhou, Lingfeng Qiao, Yulei Qin, Ke Li, Xing Sun, Xiaoyu Tan, Chao Qu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27483 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: planning-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, world-model +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27483 diff --git a/papers/items/2026-2606-27492-queenbee-planner-skill-evolving-communication-topologies-for-token-efficient-llm.md b/papers/items/2026-2606-27492-queenbee-planner-skill-evolving-communication-topologies-for-token-efficient-llm.md new file mode 100644 index 0000000..bd9fe1a --- /dev/null +++ b/papers/items/2026-2606-27492-queenbee-planner-skill-evolving-communication-topologies-for-token-efficient-llm.md @@ -0,0 +1,61 @@ +# Paper: QueenBee Planner: Skill-Evolving Communication Topologies for Token-Efficient LLM Multi-Agent Systems + +--- +type: paper +title: "QueenBee Planner: Skill-Evolving Communication Topologies for Token-Efficient LLM Multi-Agent Systems" +authors: Congjia Tian, Yuhang Yao, Jiaming Cui +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27492 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, planning, tool-use +- arXiv categories: cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27492 diff --git a/papers/items/2026-2606-27499-dmv-bench-diagnosing-long-horizon-multimodal-agents-visual-memory-with-incidenta.md b/papers/items/2026-2606-27499-dmv-bench-diagnosing-long-horizon-multimodal-agents-visual-memory-with-incidenta.md new file mode 100644 index 0000000..4c7c9b4 --- /dev/null +++ b/papers/items/2026-2606-27499-dmv-bench-diagnosing-long-horizon-multimodal-agents-visual-memory-with-incidenta.md @@ -0,0 +1,65 @@ +# Paper: DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection + +--- +type: paper +title: "DMV-Bench: Diagnosing Long-Horizon Multimodal Agents' Visual Memory with Incidental Cue Injection" +authors: Yujin Tang, Chenming Shang, Ruize Xu, Nikhil Singh +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27499 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, coding-agent, memory, planning, rag, tool-use +- arXiv categories: cs.CV, cs.AI, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27499 diff --git a/papers/items/2026-2606-27632-yuvion-llm-an-adversarially-aware-large-language-model-for-content-and-ai-safety.md b/papers/items/2026-2606-27632-yuvion-llm-an-adversarially-aware-large-language-model-for-content-and-ai-safety.md new file mode 100644 index 0000000..31dc712 --- /dev/null +++ b/papers/items/2026-2606-27632-yuvion-llm-an-adversarially-aware-large-language-model-for-content-and-ai-safety.md @@ -0,0 +1,62 @@ +# Paper: Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety + +--- +type: paper +title: "Yuvion LLM: An Adversarially-Aware Large Language Model for Content And AI Safety" +authors: Ting Ma, Xiufeng Huang, Benlei Cui, Xiaowen Xu, Shikai Qiu, Ruijie Jian, Hongxing Li, Guanghui Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27632 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, agent-safety, planning, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27632 diff --git a/papers/items/2026-2606-27806-agent-vs-parametric-world-models-hybrid-planning-for-reliable-language-agents.md b/papers/items/2026-2606-27806-agent-vs-parametric-world-models-hybrid-planning-for-reliable-language-agents.md new file mode 100644 index 0000000..7064487 --- /dev/null +++ b/papers/items/2026-2606-27806-agent-vs-parametric-world-models-hybrid-planning-for-reliable-language-agents.md @@ -0,0 +1,64 @@ +# Paper: Agent vs. Parametric World Models: Hybrid Planning for Reliable Language Agents + +--- +type: paper +title: "Agent vs. Parametric World Models: Hybrid Planning for Reliable Language Agents" +authors: Xinyuan Song, Zekun Cai +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27806 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, agent-safety, planning, rag, reasoning, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27806 diff --git a/papers/items/2026-2606-27929-when-multi-robot-systems-meet-agentic-ai-towards-embodied-collective-intelligenc.md b/papers/items/2026-2606-27929-when-multi-robot-systems-meet-agentic-ai-towards-embodied-collective-intelligenc.md new file mode 100644 index 0000000..50c673b --- /dev/null +++ b/papers/items/2026-2606-27929-when-multi-robot-systems-meet-agentic-ai-towards-embodied-collective-intelligenc.md @@ -0,0 +1,62 @@ +# Paper: When Multi-Robot Systems Meet Agentic AI:Towards Embodied Collective Intelligence + +--- +type: paper +title: "When Multi-Robot Systems Meet Agentic AI:Towards Embodied Collective Intelligence" +authors: Yuxuan Yan, Yuanyuan Jia, Qianqian Yang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27929 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, embodied-agent, memory, multi-agent, tool-use +- arXiv categories: cs.RO +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27929 diff --git a/papers/items/2026-2606-27990-advancedshellm-a-stateful-multi-agent-llm-honeypot-for-ssh-deception.md b/papers/items/2026-2606-27990-advancedshellm-a-stateful-multi-agent-llm-honeypot-for-ssh-deception.md new file mode 100644 index 0000000..d61b1ff --- /dev/null +++ b/papers/items/2026-2606-27990-advancedshellm-a-stateful-multi-agent-llm-honeypot-for-ssh-deception.md @@ -0,0 +1,61 @@ +# Paper: AdvancedShelLM: A Stateful Multi-Agent LLM Honeypot for SSH Deception + +--- +type: paper +title: "AdvancedShelLM: A Stateful Multi-Agent LLM Honeypot for SSH Deception" +authors: Muris Sladić, Eman Alibalić, Veronica Valeros, Carlos Catania, Sebastian Garcia +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.27990 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, memory, multi-agent, tool-use +- arXiv categories: cs.CR +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.27990 diff --git a/papers/items/2026-2606-28011-from-detection-to-action-using-llm-agents-for-fault-tolerant-control.md b/papers/items/2026-2606-28011-from-detection-to-action-using-llm-agents-for-fault-tolerant-control.md new file mode 100644 index 0000000..a6f99df --- /dev/null +++ b/papers/items/2026-2606-28011-from-detection-to-action-using-llm-agents-for-fault-tolerant-control.md @@ -0,0 +1,66 @@ +# Paper: From Detection to Action: Using LLM Agents for Fault-Tolerant Control + +--- +type: paper +title: "From Detection to Action: Using LLM Agents for Fault-Tolerant Control" +authors: Javal Vyas, Milapji Singh Gill, Artan Markaj, Felix Gehlhoff, Mehmet Mercangöz +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28011 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - planning + - rag + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 24 +collection_queries: agentic-ai, llm-agent, multi-agent-llm, planning-agent, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, llm-agent, multi-agent-llm, planning-agent, rag-agent +- inferred topics: agent-evaluation, agent-safety, multi-agent, planning, rag, tool-use, workflow-agent, world-model +- arXiv categories: eess.SY, cs.LG +- collection score: 24 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28011 diff --git a/papers/items/2026-2606-28061-toolprivacybench-benchmarking-purpose-bound-privacy-in-tool-using-llm-agents.md b/papers/items/2026-2606-28061-toolprivacybench-benchmarking-purpose-bound-privacy-in-tool-using-llm-agents.md new file mode 100644 index 0000000..b23bd1a --- /dev/null +++ b/papers/items/2026-2606-28061-toolprivacybench-benchmarking-purpose-bound-privacy-in-tool-using-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: ToolPrivacyBench: Benchmarking Purpose-Bound Privacy in Tool-Using LLM Agents + +--- +type: paper +title: "ToolPrivacyBench: Benchmarking Purpose-Bound Privacy in Tool-Using LLM Agents" +authors: Shijing Hu, Liang Liu, Zhu Meng, Zhicheng Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28061 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: agent-evaluation, function-calling, llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, function-calling, llm-agent, tool-use +- inferred topics: agent-evaluation, rag, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28061 diff --git a/papers/items/2026-2606-28182-llawco-learning-laws-of-cooperation-for-modeling-embodied-multi-agent-behavior.md b/papers/items/2026-2606-28182-llawco-learning-laws-of-cooperation-for-modeling-embodied-multi-agent-behavior.md new file mode 100644 index 0000000..27f5d91 --- /dev/null +++ b/papers/items/2026-2606-28182-llawco-learning-laws-of-cooperation-for-modeling-embodied-multi-agent-behavior.md @@ -0,0 +1,66 @@ +# Paper: LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior + +--- +type: paper +title: "LLawCo: Learning Laws of Cooperation for Modeling Embodied Multi-Agent Behavior" +authors: Qinhong Zhou, Chuang Gan, Anoop Cherian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28182 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - multi-agent + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CV + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, embodied-agent, multi-agent, planning, rag, reasoning +- arXiv categories: cs.LG, cs.AI, cs.CV, cs.RO +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28182 diff --git a/papers/items/2026-2606-28187-gbc-gradient-based-connections-for-optimizing-multi-agent-systems.md b/papers/items/2026-2606-28187-gbc-gradient-based-connections-for-optimizing-multi-agent-systems.md new file mode 100644 index 0000000..e076ff7 --- /dev/null +++ b/papers/items/2026-2606-28187-gbc-gradient-based-connections-for-optimizing-multi-agent-systems.md @@ -0,0 +1,60 @@ +# Paper: GBC: Gradient-Based Connections for Optimizing Multi-Agent Systems + +--- +type: paper +title: "GBC: Gradient-Based Connections for Optimizing Multi-Agent Systems" +authors: Xiaocheng Yang, Abdulrahman Alrabah, Dilek Hakkani-Tür, Gokhan Tur +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28187 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: multi-agent, rag, tool-use +- arXiv categories: cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28187 diff --git a/papers/items/2026-2606-28270-agent-native-immune-system-architecture-taxonomy-and-engineering.md b/papers/items/2026-2606-28270-agent-native-immune-system-architecture-taxonomy-and-engineering.md new file mode 100644 index 0000000..797ce57 --- /dev/null +++ b/papers/items/2026-2606-28270-agent-native-immune-system-architecture-taxonomy-and-engineering.md @@ -0,0 +1,65 @@ +# Paper: Agent-Native Immune System: Architecture, Taxonomy, and Engineering + +--- +type: paper +title: "Agent-Native Immune System: Architecture, Taxonomy, and Engineering" +authors: Bo Shen, Lifeng Chang, Tianyuan Wei, Yunpeng Li, Feng Shi, Yichen Han, Peijie Gao, Shiyi Kuang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28270 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - multi-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, agent-safety, memory, multi-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28270 diff --git a/papers/items/2026-2606-28279-agentic-hardware-design-as-repository-level-code-evolution.md b/papers/items/2026-2606-28279-agentic-hardware-design-as-repository-level-code-evolution.md new file mode 100644 index 0000000..c17d48f --- /dev/null +++ b/papers/items/2026-2606-28279-agentic-hardware-design-as-repository-level-code-evolution.md @@ -0,0 +1,60 @@ +# Paper: Agentic Hardware Design as Repository-Level Code Evolution + +--- +type: paper +title: Agentic Hardware Design as Repository-Level Code Evolution +authors: Cunxi Yu, Chenhui Deng, Nathaniel Pinckney, Brucek Khailany +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28279 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, coding-agent +- arXiv categories: cs.AR, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28279 diff --git a/papers/items/2026-2606-28349-hmars-a-hierarchical-multi-agent-memory-system-for-long-context-reasoning.md b/papers/items/2026-2606-28349-hmars-a-hierarchical-multi-agent-memory-system-for-long-context-reasoning.md new file mode 100644 index 0000000..c0c1906 --- /dev/null +++ b/papers/items/2026-2606-28349-hmars-a-hierarchical-multi-agent-memory-system-for-long-context-reasoning.md @@ -0,0 +1,64 @@ +# Paper: HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning + +--- +type: paper +title: "HMARS: A Hierarchical Multi-Agent Memory System for Long-Context Reasoning" +authors: Zeju Li, Ziyang Zheng, Yizhou Zhou, Qiang Xu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28349 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-03 +updated_at: 2026-06-03 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.IR, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28349 diff --git a/papers/items/2026-2606-28360-carolina-guide-a-multi-agent-rag-system-with-institutional-guardrails-for-academ.md b/papers/items/2026-2606-28360-carolina-guide-a-multi-agent-rag-system-with-institutional-guardrails-for-academ.md new file mode 100644 index 0000000..4a66095 --- /dev/null +++ b/papers/items/2026-2606-28360-carolina-guide-a-multi-agent-rag-system-with-institutional-guardrails-for-academ.md @@ -0,0 +1,63 @@ +# Paper: Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance + +--- +type: paper +title: "Carolina Guide: A Multi-Agent RAG System with Institutional Guardrails for Academic Policy Assistance" +authors: Ben Torsion, Jun Zhou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28360 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-11 +updated_at: 2026-06-11 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, rag +- arXiv categories: cs.IR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28360 diff --git a/papers/items/2026-2606-28374-recursive-self-evolving-agents-via-held-out-selection.md b/papers/items/2026-2606-28374-recursive-self-evolving-agents-via-held-out-selection.md new file mode 100644 index 0000000..d3601ae --- /dev/null +++ b/papers/items/2026-2606-28374-recursive-self-evolving-agents-via-held-out-selection.md @@ -0,0 +1,61 @@ +# Paper: Recursive Self-Evolving Agents via Held-Out Selection + +--- +type: paper +title: Recursive Self-Evolving Agents via Held-Out Selection +authors: Michael Nguyen, Quoc Nguyen, Paul Vuong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28374 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-17 +updated_at: 2026-06-17 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28374 diff --git a/papers/items/2026-2606-28409-evidence-driven-llm-agent-for-c-to-synthesizable-c-conversion-and-verification.md b/papers/items/2026-2606-28409-evidence-driven-llm-agent-for-c-to-synthesizable-c-conversion-and-verification.md new file mode 100644 index 0000000..b2264cc --- /dev/null +++ b/papers/items/2026-2606-28409-evidence-driven-llm-agent-for-c-to-synthesizable-c-conversion-and-verification.md @@ -0,0 +1,63 @@ +# Paper: Evidence-Driven LLM Agent for C-to-Synthesizable-C Conversion and Verification + +--- +type: paper +title: Evidence-Driven LLM Agent for C-to-Synthesizable-C Conversion and Verification +authors: Zhe Zhao, Hongbing Lang, Zhihan Xiao, Luke Ztz Hu, John Imoleayo Adebisi, Songping Mai +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28409 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - rag + - reasoning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: rag, reasoning, tool-use, workflow-agent, world-model +- arXiv categories: cs.AR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28409 diff --git a/papers/items/2026-2606-28425-tool-use-enables-undetectable-steganography-in-multi-agent-llm-systems.md b/papers/items/2026-2606-28425-tool-use-enables-undetectable-steganography-in-multi-agent-llm-systems.md new file mode 100644 index 0000000..9b1bdb7 --- /dev/null +++ b/papers/items/2026-2606-28425-tool-use-enables-undetectable-steganography-in-multi-agent-llm-systems.md @@ -0,0 +1,65 @@ +# Paper: Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems + +--- +type: paper +title: Tool Use Enables Undetectable Steganography in Multi-Agent LLM Systems +authors: Jimmy Laurence Rippin, Simon C. Marshall, David Demitri Africa, Christian Schroeder de Witt +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28425 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-25 +updated_at: 2026-06-25 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 25 +collection_queries: agentic-ai, ai-agent, autonomous-agent-llm, multi-agent-llm, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, ai-agent, autonomous-agent-llm, multi-agent-llm, tool-use +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, multi-agent, rag, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 25 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28425 diff --git a/papers/items/2026-2606-28430-building-to-the-test-coding-agents-deliver-what-you-check-not-what-you-requested.md b/papers/items/2026-2606-28430-building-to-the-test-coding-agents-deliver-what-you-check-not-what-you-requested.md new file mode 100644 index 0000000..06ab757 --- /dev/null +++ b/papers/items/2026-2606-28430-building-to-the-test-coding-agents-deliver-what-you-check-not-what-you-requested.md @@ -0,0 +1,60 @@ +# Paper: Building to the Test: Coding Agents Deliver What You Check, Not What You Requested + +--- +type: paper +title: "Building to the Test: Coding Agents Deliver What You Check, Not What You Requested" +authors: Yanuo Ma, Ben Kereopa-Yorke, Ben Schultz +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28430 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent +- arXiv categories: cs.SE, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28430 diff --git a/papers/items/2026-2606-28434-swe-mem-learning-adaptive-memory-management-for-long-horizon-coding-agents.md b/papers/items/2026-2606-28434-swe-mem-learning-adaptive-memory-management-for-long-horizon-coding-agents.md new file mode 100644 index 0000000..9886e34 --- /dev/null +++ b/papers/items/2026-2606-28434-swe-mem-learning-adaptive-memory-management-for-long-horizon-coding-agents.md @@ -0,0 +1,63 @@ +# Paper: SWE-MeM: Learning Adaptive Memory Management for Long-Horizon Coding Agents + +--- +type: paper +title: "SWE-MeM: Learning Adaptive Memory Management for Long-Horizon Coding Agents" +authors: Shuzheng Gao, Wenhao Zeng, Zhaojian Yu, Jianqiao Wangni, Chaozheng Wang, Kai Cai, Shilin He, Michael R. Lyu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28434 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - coding-agent + - memory + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: coding-agent, memory, planning, tool-use, workflow-agent +- arXiv categories: cs.SE, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28434 diff --git a/papers/items/2026-2606-28436-dockerless-environment-free-program-verifier-for-coding-agents.md b/papers/items/2026-2606-28436-dockerless-environment-free-program-verifier-for-coding-agents.md new file mode 100644 index 0000000..09453e2 --- /dev/null +++ b/papers/items/2026-2606-28436-dockerless-environment-free-program-verifier-for-coding-agents.md @@ -0,0 +1,61 @@ +# Paper: Dockerless: Environment-Free Program Verifier for Coding Agents + +--- +type: paper +title: "Dockerless: Environment-Free Program Verifier for Coding Agents" +authors: Wenhao Zeng, Yuling Shi, Xiaodong Gu, Chao Hu, Chaofan Wang, Yuhao Cui, Hongting Zhou, Mengnan Qi, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28436 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, reasoning +- arXiv categories: cs.SE, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28436 diff --git a/papers/items/2026-2606-28450-llm-agents-security-duality-a-comprehensive-survey-of-self-security-and-empowere.md b/papers/items/2026-2606-28450-llm-agents-security-duality-a-comprehensive-survey-of-self-security-and-empowere.md new file mode 100644 index 0000000..f611051 --- /dev/null +++ b/papers/items/2026-2606-28450-llm-agents-security-duality-a-comprehensive-survey-of-self-security-and-empowere.md @@ -0,0 +1,61 @@ +# Paper: LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity + +--- +type: paper +title: "LLM agents security duality: a comprehensive survey of self-security and empowered cybersecurity" +authors: Yiwei Xu, Yong Zhuang, Xuanming Liu, Tian Zhang, Bowen Xiao, Xiaoyang Xu, Delong Jiang, Juan Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28450 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-safety, llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, llm-agent, tool-use +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28450 diff --git a/papers/items/2026-2606-28456-is-lying-an-emergent-behaviour-in-llms-evidence-from-gaslighting-ai-agents-in-a-.md b/papers/items/2026-2606-28456-is-lying-an-emergent-behaviour-in-llms-evidence-from-gaslighting-ai-agents-in-a-.md new file mode 100644 index 0000000..2642a06 --- /dev/null +++ b/papers/items/2026-2606-28456-is-lying-an-emergent-behaviour-in-llms-evidence-from-gaslighting-ai-agents-in-a-.md @@ -0,0 +1,61 @@ +# Paper: Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game + +--- +type: paper +title: Is Lying an Emergent Behaviour in LLMs? Evidence from Gaslighting AI agents in a Sustainability Game +authors: Subhendu Bhandary, Federico Carucci, Christos Charalambous, Francesca Dilisante, Ksenia Dvorkina, Anna Garbo, Jiaqi Liang, Riccardo Vasellini, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28456 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-safety + - memory + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: ai-agent, llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, llm-agent, multi-agent-llm +- inferred topics: agent-safety, memory, multi-agent +- arXiv categories: cs.MA, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28456 diff --git a/papers/items/2026-2606-28467-an-agentic-ai-pipeline-for-appliance-level-energy-anomaly-detection-and-llm-driv.md b/papers/items/2026-2606-28467-an-agentic-ai-pipeline-for-appliance-level-energy-anomaly-detection-and-llm-driv.md new file mode 100644 index 0000000..b1ac996 --- /dev/null +++ b/papers/items/2026-2606-28467-an-agentic-ai-pipeline-for-appliance-level-energy-anomaly-detection-and-llm-driv.md @@ -0,0 +1,65 @@ +# Paper: An Agentic AI Pipeline for Appliance-Level Energy Anomaly Detection and LLM-Driven Recommendations + +--- +type: paper +title: An Agentic AI Pipeline for Appliance-Level Energy Anomaly Detection and LLM-Driven Recommendations +authors: Dihia Falouz, Aida Douaibia, Amine Bechar, Youssef Elmir, Abbes Amira, Adel Oulefki +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28467 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agentic-ai, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, rag-agent +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28467 diff --git a/papers/items/2026-2606-28480-tua-bench-a-benchmark-for-general-purpose-terminal-use-agents.md b/papers/items/2026-2606-28480-tua-bench-a-benchmark-for-general-purpose-terminal-use-agents.md new file mode 100644 index 0000000..34afe46 --- /dev/null +++ b/papers/items/2026-2606-28480-tua-bench-a-benchmark-for-general-purpose-terminal-use-agents.md @@ -0,0 +1,63 @@ +# Paper: TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents + +--- +type: paper +title: "TUA-Bench: A Benchmark for General-Purpose Terminal-Use Agents" +authors: Shoufa Chen, Luyuan Wang, Xuan Yang, Zhiheng Liu, Yuren Cong, Yuanfeng Ji, Feiyan Zhou, Xiaohui Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28480 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, reasoning, workflow-agent +- arXiv categories: cs.SE, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28480 diff --git a/papers/items/2026-2606-28570-digitizing-coaching-intelligence-an-agentic-framework-for-holistic-athlete-profi.md b/papers/items/2026-2606-28570-digitizing-coaching-intelligence-an-agentic-framework-for-holistic-athlete-profi.md new file mode 100644 index 0000000..56b749f --- /dev/null +++ b/papers/items/2026-2606-28570-digitizing-coaching-intelligence-an-agentic-framework-for-holistic-athlete-profi.md @@ -0,0 +1,64 @@ +# Paper: Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG + +--- +type: paper +title: "Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG" +authors: Deep Ghosal, Ishani Sen, Wazib Ansar, Amlan Chakrabarti +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28570 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-26 +updated_at: 2026-06-26 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm, rag-agent +- inferred topics: agent-evaluation, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.CV, cs.AI, cs.MA +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28570 diff --git a/papers/items/2026-2606-28666-why-trust-your-agent-empirical-security-gains-from-trism-guided-agentic-workflow.md b/papers/items/2026-2606-28666-why-trust-your-agent-empirical-security-gains-from-trism-guided-agentic-workflow.md new file mode 100644 index 0000000..5435a3c --- /dev/null +++ b/papers/items/2026-2606-28666-why-trust-your-agent-empirical-security-gains-from-trism-guided-agentic-workflow.md @@ -0,0 +1,64 @@ +# Paper: Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare + +--- +type: paper +title: Why Trust Your Agent? Empirical Security Gains from TRiSM-Guided Agentic Workflows in Healthcare +authors: Liam Kearns +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28666 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agentic-ai, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, rag-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, rag, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28666 diff --git a/papers/items/2026-2606-28679-capability-gates-are-not-authorization-confused-deputy-failures-in-llm-agent-fra.md b/papers/items/2026-2606-28679-capability-gates-are-not-authorization-confused-deputy-failures-in-llm-agent-fra.md new file mode 100644 index 0000000..76eb9de --- /dev/null +++ b/papers/items/2026-2606-28679-capability-gates-are-not-authorization-confused-deputy-failures-in-llm-agent-fra.md @@ -0,0 +1,60 @@ +# Paper: Capability Gates Are Not Authorization: Confused-Deputy Failures in LLM Agent Frameworks + +--- +type: paper +title: "Capability Gates Are Not Authorization: Confused-Deputy Failures in LLM Agent Frameworks" +authors: David Mellafe Zuvic +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28679 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, tool-use +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28679 diff --git a/papers/items/2026-2606-28692-an-ai-agent-for-treatment-reasoning-over-a-biomedical-tool-universe.md b/papers/items/2026-2606-28692-an-ai-agent-for-treatment-reasoning-over-a-biomedical-tool-universe.md new file mode 100644 index 0000000..0059d25 --- /dev/null +++ b/papers/items/2026-2606-28692-an-ai-agent-for-treatment-reasoning-over-a-biomedical-tool-universe.md @@ -0,0 +1,61 @@ +# Paper: An AI agent for treatment reasoning over a biomedical tool universe + +--- +type: paper +title: An AI agent for treatment reasoning over a biomedical tool universe +authors: Shanghua Gao, Ayush Noori, Richard Zhu, Curtis Ginder, Zhenglun Kong, Xiaorui Su, Justin Kauffman, Benjamin S. Glicksberg, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28692 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: ai-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, tool-use +- inferred topics: agent-evaluation, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28692 diff --git a/papers/items/2026-2606-28733-agentic-abstention-do-agents-know-when-to-stop-instead-of-act.md b/papers/items/2026-2606-28733-agentic-abstention-do-agents-know-when-to-stop-instead-of-act.md new file mode 100644 index 0000000..0978965 --- /dev/null +++ b/papers/items/2026-2606-28733-agentic-abstention-do-agents-know-when-to-stop-instead-of-act.md @@ -0,0 +1,60 @@ +# Paper: Agentic Abstention: Do Agents Know When to Stop Instead of Act? + +--- +type: paper +title: "Agentic Abstention: Do Agents Know When to Stop Instead of Act?" +authors: Han Luo, Bingbing Wen, Lucy Lu Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28733 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28733 diff --git a/papers/items/2026-2606-28739-agent-safety-is-action-alignment.md b/papers/items/2026-2606-28739-agent-safety-is-action-alignment.md new file mode 100644 index 0000000..e2030ce --- /dev/null +++ b/papers/items/2026-2606-28739-agent-safety-is-action-alignment.md @@ -0,0 +1,60 @@ +# Paper: Agent Safety Is Action Alignment + +--- +type: paper +title: Agent Safety Is Action Alignment +authors: Shawn Li, Yue Zhao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28739 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28739 diff --git a/papers/items/2026-2606-28781-hyphaedb-a-living-knowledge-topology-for-agent-first-memory.md b/papers/items/2026-2606-28781-hyphaedb-a-living-knowledge-topology-for-agent-first-memory.md new file mode 100644 index 0000000..ead4de7 --- /dev/null +++ b/papers/items/2026-2606-28781-hyphaedb-a-living-knowledge-topology-for-agent-first-memory.md @@ -0,0 +1,63 @@ +# Paper: HyphaeDB: A Living Knowledge Topology for Agent-First Memory + +--- +type: paper +title: "HyphaeDB: A Living Knowledge Topology for Agent-First Memory" +authors: Krishna Halaharvi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28781 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - coding-agent + - memory + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory, agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, agentic-ai +- inferred topics: coding-agent, memory, multi-agent, rag, tool-use +- arXiv categories: cs.AI, cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28781 diff --git a/papers/items/2026-2606-28791-from-determinism-to-delegation-ai-native-software-engineering-and-the-evolution-.md b/papers/items/2026-2606-28791-from-determinism-to-delegation-ai-native-software-engineering-and-the-evolution-.md new file mode 100644 index 0000000..5175247 --- /dev/null +++ b/papers/items/2026-2606-28791-from-determinism-to-delegation-ai-native-software-engineering-and-the-evolution-.md @@ -0,0 +1,65 @@ +# Paper: From Determinism to Delegation: AI-Native Software Engineering and the Evolution of the Agentic Engineer + +--- +type: paper +title: "From Determinism to Delegation: AI-Native Software Engineering and the Evolution of the Agentic Engineer" +authors: Mamdouh Alenezi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28791 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - multi-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 23 +collection_queries: agentic-ai, autonomous-agent-llm, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, autonomous-agent-llm, tool-use +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, multi-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.SE +- collection score: 23 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28791 diff --git a/papers/items/2026-2606-28839-the-contagion-tensor-a-framework-for-measuring-output-distribution-coupling-in-m.md b/papers/items/2026-2606-28839-the-contagion-tensor-a-framework-for-measuring-output-distribution-coupling-in-m.md new file mode 100644 index 0000000..eaae3ff --- /dev/null +++ b/papers/items/2026-2606-28839-the-contagion-tensor-a-framework-for-measuring-output-distribution-coupling-in-m.md @@ -0,0 +1,62 @@ +# Paper: The Contagion Tensor: A Framework for Measuring Output-Distribution Coupling in Multi-Agent LLM Systems -- and Auditing the Claims It Enables + +--- +type: paper +title: "The Contagion Tensor: A Framework for Measuring Output-Distribution Coupling in Multi-Agent LLM Systems -- and Auditing the Claims It Enables" +authors: Zewen Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28839 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, computer-use, multi-agent, tool-use, world-model +- arXiv categories: cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28839 diff --git a/papers/items/2026-2606-28841-lamp-lean-based-agentic-framework-with-mcp-and-proof-repair.md b/papers/items/2026-2606-28841-lamp-lean-based-agentic-framework-with-mcp-and-proof-repair.md new file mode 100644 index 0000000..9db8405 --- /dev/null +++ b/papers/items/2026-2606-28841-lamp-lean-based-agentic-framework-with-mcp-and-proof-repair.md @@ -0,0 +1,64 @@ +# Paper: LAMP: Lean-based Agentic framework with MCP and Proof Repair + +--- +type: paper +title: "LAMP: Lean-based Agentic framework with MCP and Proof Repair" +authors: Santhana Srinivasan R, Maithilee Patawar +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28841 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - coding-agent + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LO + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: coding-agent, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.LO, cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28841 diff --git a/papers/items/2026-2606-28896-a-task-driven-and-quality-assured-agent-framework-for-sar-data-generation.md b/papers/items/2026-2606-28896-a-task-driven-and-quality-assured-agent-framework-for-sar-data-generation.md new file mode 100644 index 0000000..e4bfa97 --- /dev/null +++ b/papers/items/2026-2606-28896-a-task-driven-and-quality-assured-agent-framework-for-sar-data-generation.md @@ -0,0 +1,63 @@ +# Paper: A Task-Driven and Quality-Assured Agent Framework for SAR Data Generation + +--- +type: paper +title: A Task-Driven and Quality-Assured Agent Framework for SAR Data Generation +authors: Xuanting Wu, Fan Zhanga, Fei Ma, Ling Guan, Guochun Ma, Yongsheng Zhou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28896 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.IV + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, planning, rag, workflow-agent +- arXiv categories: eess.IV, cs.AI, cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28896 diff --git a/papers/items/2026-2606-28925-multi-agent-routing-as-set-valued-prediction-a-wildchat-benchmark-and-cost-aware.md b/papers/items/2026-2606-28925-multi-agent-routing-as-set-valued-prediction-a-wildchat-benchmark-and-cost-aware.md new file mode 100644 index 0000000..27c83d5 --- /dev/null +++ b/papers/items/2026-2606-28925-multi-agent-routing-as-set-valued-prediction-a-wildchat-benchmark-and-cost-aware.md @@ -0,0 +1,65 @@ +# Paper: Multi-Agent Routing as Set-Valued Prediction: A WildChat Benchmark and Cost-Aware Evaluation + +--- +type: paper +title: "Multi-Agent Routing as Set-Valued Prediction: A WildChat Benchmark and Cost-Aware Evaluation" +authors: Ananto Nayan Bala, Faisal Muhammad Shah +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28925 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.IR + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, rag, tool-use, world-model +- arXiv categories: cs.LG, cs.AI, cs.IR, cs.MA +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28925 diff --git a/papers/items/2026-2606-28958-when-latent-agents-lie-kv-cache-integrity-in-multi-agent-llm-collaboration.md b/papers/items/2026-2606-28958-when-latent-agents-lie-kv-cache-integrity-in-multi-agent-llm-collaboration.md new file mode 100644 index 0000000..0f37f1e --- /dev/null +++ b/papers/items/2026-2606-28958-when-latent-agents-lie-kv-cache-integrity-in-multi-agent-llm-collaboration.md @@ -0,0 +1,61 @@ +# Paper: When Latent Agents Lie: KV-Cache Integrity in Multi-Agent LLM Collaboration + +--- +type: paper +title: "When Latent Agents Lie: KV-Cache Integrity in Multi-Agent LLM Collaboration" +authors: Luís Brito, Carlos Baquero +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.28958 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-safety + - memory + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-safety, memory, multi-agent, reasoning +- arXiv categories: cs.MA +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.28958 diff --git a/papers/items/2026-2606-29014-customized-generative-ai-agent-for-transportation-engineering-practice-a-develop.md b/papers/items/2026-2606-29014-customized-generative-ai-agent-for-transportation-engineering-practice-a-develop.md new file mode 100644 index 0000000..0e50bca --- /dev/null +++ b/papers/items/2026-2606-29014-customized-generative-ai-agent-for-transportation-engineering-practice-a-develop.md @@ -0,0 +1,63 @@ +# Paper: Customized Generative AI Agent for Transportation Engineering Practice: A Development and Continued Pre-training Guideline + +--- +type: paper +title: "Customized Generative AI Agent for Transportation Engineering Practice: A Development and Continued Pre-training Guideline" +authors: Dianwei Chen, Yuan-Zheng Lei, Zifan Zhang, Yuchen Liu, Xianfeng Yang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29014 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - planning + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.DL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, planning, reasoning +- arXiv categories: cs.AI, cs.DL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29014 diff --git a/papers/items/2026-2606-29026-preventing-error-propagation-in-multi-agent-ai-through-runtime-monitoring.md b/papers/items/2026-2606-29026-preventing-error-propagation-in-multi-agent-ai-through-runtime-monitoring.md new file mode 100644 index 0000000..0e216a7 --- /dev/null +++ b/papers/items/2026-2606-29026-preventing-error-propagation-in-multi-agent-ai-through-runtime-monitoring.md @@ -0,0 +1,62 @@ +# Paper: Preventing Error Propagation in Multi-Agent AI through Runtime Monitoring + +--- +type: paper +title: Preventing Error Propagation in Multi-Agent AI through Runtime Monitoring +authors: Shahnewaz Karim Sakib, Anindya Bijoy Das +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29026 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.ET +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, agent-safety, multi-agent, reasoning +- arXiv categories: cs.AI, cs.ET +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29026 diff --git a/papers/items/2026-2606-29030-memory-as-an-attack-surface-in-llm-agents-a-study-on-multiple-choice-question-an.md b/papers/items/2026-2606-29030-memory-as-an-attack-surface-in-llm-agents-a-study-on-multiple-choice-question-an.md new file mode 100644 index 0000000..9adb5ad --- /dev/null +++ b/papers/items/2026-2606-29030-memory-as-an-attack-surface-in-llm-agents-a-study-on-multiple-choice-question-an.md @@ -0,0 +1,60 @@ +# Paper: Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering + +--- +type: paper +title: "Memory as an Attack Surface in LLM Agents: A Study on Multiple-Choice Question Answering" +authors: Shahnewaz Karim Sakib, Anindya Bijoy Das +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29030 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.ET +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: ai-agent, llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, llm-agent, tool-use +- inferred topics: memory, tool-use +- arXiv categories: cs.AI, cs.ET +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29030 diff --git a/papers/items/2026-2606-29116-characterizing-large-language-model-agentic-workflows-a-study-on-n8n-ecosystem.md b/papers/items/2026-2606-29116-characterizing-large-language-model-agentic-workflows-a-study-on-n8n-ecosystem.md new file mode 100644 index 0000000..f7b8465 --- /dev/null +++ b/papers/items/2026-2606-29116-characterizing-large-language-model-agentic-workflows-a-study-on-n8n-ecosystem.md @@ -0,0 +1,62 @@ +# Paper: Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem + +--- +type: paper +title: "Characterizing Large Language Model Agentic Workflows: A Study on N8n Ecosystem" +authors: Yutian Tang, Yuming Zhou, Huaming Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29116 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-27 +updated_at: 2026-06-27 +status: queued +relevance: high +topics: + - agent-safety + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agentic-ai, llm-agent, planning-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, llm-agent, planning-agent, tool-use +- inferred topics: agent-safety, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29116 diff --git a/papers/items/2026-2606-29142-agent-security-meets-regulatory-reality-a-practitioner-systematization-of-autono.md b/papers/items/2026-2606-29142-agent-security-meets-regulatory-reality-a-practitioner-systematization-of-autono.md new file mode 100644 index 0000000..35a04a0 --- /dev/null +++ b/papers/items/2026-2606-29142-agent-security-meets-regulatory-reality-a-practitioner-systematization-of-autono.md @@ -0,0 +1,62 @@ +# Paper: Agent Security Meets Regulatory Reality -- A Practitioner Systematization of Autonomous-Agent Threats and Controls in Regulated Financial Systems + +--- +type: paper +title: Agent Security Meets Regulatory Reality -- A Practitioner Systematization of Autonomous-Agent Threats and Controls in Regulated Financial Systems +authors: Krishna Mohan, Guda Nagavenkata Srinivasa +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29142 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety, autonomous-agent-llm, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, autonomous-agent-llm, rag-agent +- inferred topics: agent-safety, computer-use, rag, tool-use +- arXiv categories: cs.CY, cs.SE +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29142 diff --git a/papers/items/2026-2606-29178-selective-memory-retention-for-long-horizon-llm-agents.md b/papers/items/2026-2606-29178-selective-memory-retention-for-long-horizon-llm-agents.md new file mode 100644 index 0000000..b54aec2 --- /dev/null +++ b/papers/items/2026-2606-29178-selective-memory-retention-for-long-horizon-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: Selective Memory Retention for Long-Horizon LLM Agents + +--- +type: paper +title: Selective Memory Retention for Long-Horizon LLM Agents +authors: Pranath Reddy +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29178 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, memory, planning +- arXiv categories: cs.AI, cs.CL, cs.LG +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29178 diff --git a/papers/items/2026-2606-29193-a-multi-dataset-benchmark-for-evaluating-llm-agents-in-microservice-failure-diag.md b/papers/items/2026-2606-29193-a-multi-dataset-benchmark-for-evaluating-llm-agents-in-microservice-failure-diag.md new file mode 100644 index 0000000..ea0ab1e --- /dev/null +++ b/papers/items/2026-2606-29193-a-multi-dataset-benchmark-for-evaluating-llm-agents-in-microservice-failure-diag.md @@ -0,0 +1,62 @@ +# Paper: A Multi-Dataset Benchmark for Evaluating LLM Agents in Microservice Failure Diagnosis + +--- +type: paper +title: A Multi-Dataset Benchmark for Evaluating LLM Agents in Microservice Failure Diagnosis +authors: Yuanhong Cai, Xiaohui Nie, Kanglin Yin, Changhua Pei, Yongqian Sun, Shenglin Zhang, Haibin Liu, Guiyang Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29193 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, rag, reasoning, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29193 diff --git a/papers/items/2026-2606-29225-policyguard-a-dialogue-grounded-sub-agent-verifier-for-policy-adherence-in-llm-a.md b/papers/items/2026-2606-29225-policyguard-a-dialogue-grounded-sub-agent-verifier-for-policy-adherence-in-llm-a.md new file mode 100644 index 0000000..3ff0837 --- /dev/null +++ b/papers/items/2026-2606-29225-policyguard-a-dialogue-grounded-sub-agent-verifier-for-policy-adherence-in-llm-a.md @@ -0,0 +1,62 @@ +# Paper: PolicyGuard: A Dialogue-Grounded Sub-Agent Verifier for Policy Adherence in LLM Agents + +--- +type: paper +title: "PolicyGuard: A Dialogue-Grounded Sub-Agent Verifier for Policy Adherence in LLM Agents" +authors: Seongjae Kang, Taehyung Yu, Sung Ju Hwang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29225 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - computer-use + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: computer-use, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29225 diff --git a/papers/items/2026-2606-29270-minority-sentinel-when-to-overturn-majority-voting-in-multi-agent-llm-debates.md b/papers/items/2026-2606-29270-minority-sentinel-when-to-overturn-majority-voting-in-multi-agent-llm-debates.md new file mode 100644 index 0000000..4de6894 --- /dev/null +++ b/papers/items/2026-2606-29270-minority-sentinel-when-to-overturn-majority-voting-in-multi-agent-llm-debates.md @@ -0,0 +1,61 @@ +# Paper: Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates + +--- +type: paper +title: "Minority Sentinel: When to Overturn Majority Voting in Multi-Agent LLM Debates" +authors: Chuan He, Zebin Chen, Zhengyi Yang, Shaobo Qiao, Mingchen Ju, Jiate Liu, Dong Wen, Guanfeng Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29270 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, reasoning +- arXiv categories: cs.MA +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29270 diff --git a/papers/items/2026-2606-29315-hierarchical-experimentalist-agents.md b/papers/items/2026-2606-29315-hierarchical-experimentalist-agents.md new file mode 100644 index 0000000..fa2ff30 --- /dev/null +++ b/papers/items/2026-2606-29315-hierarchical-experimentalist-agents.md @@ -0,0 +1,64 @@ +# Paper: Hierarchical Experimentalist Agents + +--- +type: paper +title: Hierarchical Experimentalist Agents +authors: Abhranil Chandra, Sankaran Vaidyanathan, Utsav Dhanuka, Varun Gandhi, Scott Niekum +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29315 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use, world-model +- arXiv categories: cs.AI, cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29315 diff --git a/papers/items/2026-2606-29354-when-llms-develop-languages-symbolic-communication-for-efficient-multi-agent-rea.md b/papers/items/2026-2606-29354-when-llms-develop-languages-symbolic-communication-for-efficient-multi-agent-rea.md new file mode 100644 index 0000000..5198728 --- /dev/null +++ b/papers/items/2026-2606-29354-when-llms-develop-languages-symbolic-communication-for-efficient-multi-agent-rea.md @@ -0,0 +1,61 @@ +# Paper: When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning + +--- +type: paper +title: "When LLMs Develop Languages: Symbolic Communication for Efficient Multi-Agent Reasoning" +authors: Zhengqi Pei, Qingming Huang, Shuhui Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29354 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.NE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, reasoning +- arXiv categories: cs.AI, cs.NE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29354 diff --git a/papers/items/2026-2606-29445-bridging-videoqa-and-video-guided-agentic-tasks-via-generalized-keyframe-extract.md b/papers/items/2026-2606-29445-bridging-videoqa-and-video-guided-agentic-tasks-via-generalized-keyframe-extract.md new file mode 100644 index 0000000..e8cc255 --- /dev/null +++ b/papers/items/2026-2606-29445-bridging-videoqa-and-video-guided-agentic-tasks-via-generalized-keyframe-extract.md @@ -0,0 +1,62 @@ +# Paper: Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction + +--- +type: paper +title: Bridging VideoQA and Video-Guided Agentic Tasks via Generalized Keyframe Extraction +authors: Sunqi Fan, Qingle Liu, Runqi Yin, Meng-Hao Guo, Shuojin Yang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29445 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use +- arXiv categories: cs.CV, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29445 diff --git a/papers/items/2026-2606-29495-cognitive-world-models-for-process-level-social-influence-evaluation.md b/papers/items/2026-2606-29495-cognitive-world-models-for-process-level-social-influence-evaluation.md new file mode 100644 index 0000000..4760636 --- /dev/null +++ b/papers/items/2026-2606-29495-cognitive-world-models-for-process-level-social-influence-evaluation.md @@ -0,0 +1,61 @@ +# Paper: Cognitive World Models for Process-Level Social Influence Evaluation + +--- +type: paper +title: Cognitive World Models for Process-Level Social Influence Evaluation +authors: Minghui Ma, Bin Guo, Han Wang, Mengqi Chen, Jingqi Liu, Yan Liu, Zhiwen Yu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29495 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, computer-use, multi-agent, world-model +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29495 diff --git a/papers/items/2026-2606-29537-osworld2-0-benchmarking-computer-use-agents-on-long-horizon-real-world-tasks.md b/papers/items/2026-2606-29537-osworld2-0-benchmarking-computer-use-agents-on-long-horizon-real-world-tasks.md new file mode 100644 index 0000000..d100c38 --- /dev/null +++ b/papers/items/2026-2606-29537-osworld2-0-benchmarking-computer-use-agents-on-long-horizon-real-world-tasks.md @@ -0,0 +1,67 @@ +# Paper: OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks + +--- +type: paper +title: "OSWorld2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks" +authors: Mengqi Yuan, Zilong Zhou, Xinzhuang Xiong, Weiming Wu, Jiayang Sun, Jiamin Song, Kaiqian Cui, Bowen Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29537 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - memory + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, memory, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29537 diff --git a/papers/items/2026-2606-29654-budgeted-act-or-defer-multi-agent-llm-deliberation-with-local-reliability-bounds.md b/papers/items/2026-2606-29654-budgeted-act-or-defer-multi-agent-llm-deliberation-with-local-reliability-bounds.md new file mode 100644 index 0000000..6b6d306 --- /dev/null +++ b/papers/items/2026-2606-29654-budgeted-act-or-defer-multi-agent-llm-deliberation-with-local-reliability-bounds.md @@ -0,0 +1,64 @@ +# Paper: Budgeted Act-or-Defer Multi-Agent LLM Deliberation with Local Reliability Bounds + +--- +type: paper +title: Budgeted Act-or-Defer Multi-Agent LLM Deliberation with Local Reliability Bounds +authors: Mengdie Flora Wang, Haochen Xie, Guanghui Wang, Devin Zhang, Jae Oh Woo +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29654 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.MA +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29654 diff --git a/papers/items/2026-2606-29719-a-diagnostic-framework-and-multi-evaluator-audit-of-evaluator-driven-preference-.md b/papers/items/2026-2606-29719-a-diagnostic-framework-and-multi-evaluator-audit-of-evaluator-driven-preference-.md new file mode 100644 index 0000000..8fa118e --- /dev/null +++ b/papers/items/2026-2606-29719-a-diagnostic-framework-and-multi-evaluator-audit-of-evaluator-driven-preference-.md @@ -0,0 +1,60 @@ +# Paper: A Diagnostic Framework and Multi-Evaluator Audit of Evaluator-Driven Preference Dynamics in Self-Adapting LLM Agents + +--- +type: paper +title: A Diagnostic Framework and Multi-Evaluator Audit of Evaluator-Driven Preference Dynamics in Self-Adapting LLM Agents +authors: Liu Zewen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29719 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, rag +- arXiv categories: cs.LG, cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29719 diff --git a/papers/items/2026-2606-29742-microagent-context-augmented-multi-agent-framework-for-automatic-microservice-de.md b/papers/items/2026-2606-29742-microagent-context-augmented-multi-agent-framework-for-automatic-microservice-de.md new file mode 100644 index 0000000..d473242 --- /dev/null +++ b/papers/items/2026-2606-29742-microagent-context-augmented-multi-agent-framework-for-automatic-microservice-de.md @@ -0,0 +1,63 @@ +# Paper: MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition + +--- +type: paper +title: "MicroAgent: Context-Augmented Multi-Agent Framework for Automatic Microservice Decomposition" +authors: Zishan Su, Junjie Huang, Shiwen Shan, Xingyan Chen, Hui Zeng, Yuxin Su, Yanlin Wang, Michael R. Lyu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29742 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, computer-use, multi-agent, rag, tool-use +- arXiv categories: cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29742 diff --git a/papers/items/2026-2606-29745-echo-learning-epistemically-adaptive-language-agents-with-turn-level-credit.md b/papers/items/2026-2606-29745-echo-learning-epistemically-adaptive-language-agents-with-turn-level-credit.md new file mode 100644 index 0000000..a3aed8f --- /dev/null +++ b/papers/items/2026-2606-29745-echo-learning-epistemically-adaptive-language-agents-with-turn-level-credit.md @@ -0,0 +1,61 @@ +# Paper: ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit + +--- +type: paper +title: "ECHO: Learning Epistemically Adaptive Language Agents with Turn-Level Credit" +authors: Abhijnan Nath, Nikhil Krishnaswamy +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29745 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, rag, reasoning, tool-use +- arXiv categories: cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29745 diff --git a/papers/items/2026-2606-29746-deepmed-search-an-open-source-agentic-platform-for-medical-deep-research-with-in.md b/papers/items/2026-2606-29746-deepmed-search-an-open-source-agentic-platform-for-medical-deep-research-with-in.md new file mode 100644 index 0000000..156f7ed --- /dev/null +++ b/papers/items/2026-2606-29746-deepmed-search-an-open-source-agentic-platform-for-medical-deep-research-with-in.md @@ -0,0 +1,63 @@ +# Paper: DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification + +--- +type: paper +title: "DEEPMED Search: An Open-Source Agentic Platform for Medical Deep Research with Introspective Verification" +authors: Maolin Liu, Fanyu Xu, Ruoqing Xu, Jiahang Zhang, Hao Wang, Rui Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29746 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - computer-use + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: computer-use, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.HC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29746 diff --git a/papers/items/2026-2606-29762-do-recommendation-algorithms-work-when-users-are-llm-agents-a-case-study-on-molt.md b/papers/items/2026-2606-29762-do-recommendation-algorithms-work-when-users-are-llm-agents-a-case-study-on-molt.md new file mode 100644 index 0000000..c39e53f --- /dev/null +++ b/papers/items/2026-2606-29762-do-recommendation-algorithms-work-when-users-are-llm-agents-a-case-study-on-molt.md @@ -0,0 +1,59 @@ +# Paper: Do Recommendation Algorithms Work When Users Are LLM Agents? A Case Study on Moltbook + +--- +type: paper +title: Do Recommendation Algorithms Work When Users Are LLM Agents? A Case Study on Moltbook +authors: Daming Li, Simeng Han, Jialu Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29762 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: ai-agent, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, llm-agent +- inferred topics: agent-evaluation, rag +- arXiv categories: cs.IR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29762 diff --git a/papers/items/2026-2606-29771-clqt-a-closed-loop-cost-aware-strategy-consistent-benchmark-for-diagnostic-evalu.md b/papers/items/2026-2606-29771-clqt-a-closed-loop-cost-aware-strategy-consistent-benchmark-for-diagnostic-evalu.md new file mode 100644 index 0000000..6c3ad95 --- /dev/null +++ b/papers/items/2026-2606-29771-clqt-a-closed-loop-cost-aware-strategy-consistent-benchmark-for-diagnostic-evalu.md @@ -0,0 +1,64 @@ +# Paper: CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents + +--- +type: paper +title: "CLQT: A Closed-Loop, Cost-Aware, Strategy-Consistent Benchmark for Diagnostic Evaluation of LLM Portfolio-Management Agents" +authors: Bo Qu, Mingguang Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29771 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG + - q-fin.CP + - q-fin.PM +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, memory, reasoning, tool-use +- arXiv categories: cs.AI, cs.LG, q-fin.CP, q-fin.PM +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29771 diff --git a/papers/items/2026-2606-29774-analytic-concept-centric-memory-for-agentic-embodied-manipulation.md b/papers/items/2026-2606-29774-analytic-concept-centric-memory-for-agentic-embodied-manipulation.md new file mode 100644 index 0000000..3fe8066 --- /dev/null +++ b/papers/items/2026-2606-29774-analytic-concept-centric-memory-for-agentic-embodied-manipulation.md @@ -0,0 +1,64 @@ +# Paper: Analytic Concept-Centric Memory for Agentic Embodied Manipulation + +--- +type: paper +title: Analytic Concept-Centric Memory for Agentic Embodied Manipulation +authors: Mingyang Sun, Xiujian Liang, Jiude Wei, Qichen He, Donglin Wang, Cewu Lu, Jianhua Sun +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29774 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, embodied-agent, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.RO +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29774 diff --git a/papers/items/2026-2606-29778-mandol-an-agglomerative-agent-memory-system-for-long-term-conversations.md b/papers/items/2026-2606-29778-mandol-an-agglomerative-agent-memory-system-for-long-term-conversations.md new file mode 100644 index 0000000..d740c45 --- /dev/null +++ b/papers/items/2026-2606-29778-mandol-an-agglomerative-agent-memory-system-for-long-term-conversations.md @@ -0,0 +1,63 @@ +# Paper: Mandol: An Agglomerative Agent Memory System for Long-Term Conversations + +--- +type: paper +title: "Mandol: An Agglomerative Agent Memory System for Long-Term Conversations" +authors: Yuhan Zhang, Zhiyuan Guo, Ziheng Zeng, Wei Wang, Wentao Wu, Lijie Xu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29778 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DB + - cs.AI + - cs.CL + - cs.IR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, rag-agent +- inferred topics: agent-evaluation, memory, rag +- arXiv categories: cs.DB, cs.AI, cs.CL, cs.IR +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29778 diff --git a/papers/items/2026-2606-29788-memleak-diagnosing-information-leaks-in-multimodal-agent-memory.md b/papers/items/2026-2606-29788-memleak-diagnosing-information-leaks-in-multimodal-agent-memory.md new file mode 100644 index 0000000..3a7ed0c --- /dev/null +++ b/papers/items/2026-2606-29788-memleak-diagnosing-information-leaks-in-multimodal-agent-memory.md @@ -0,0 +1,59 @@ +# Paper: MemLeak: Diagnosing Information Leaks in Multimodal Agent Memory + +--- +type: paper +title: "MemLeak: Diagnosing Information Leaks in Multimodal Agent Memory" +authors: Kuan Wang, Chao Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29788 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-memory, ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, ai-agent +- inferred topics: agent-evaluation, memory +- arXiv categories: cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29788 diff --git a/papers/items/2026-2606-29824-neural-procedural-memory-empowering-llm-agents-with-implicit-activation-steering.md b/papers/items/2026-2606-29824-neural-procedural-memory-empowering-llm-agents-with-implicit-activation-steering.md new file mode 100644 index 0000000..80ab2bc --- /dev/null +++ b/papers/items/2026-2606-29824-neural-procedural-memory-empowering-llm-agents-with-implicit-activation-steering.md @@ -0,0 +1,64 @@ +# Paper: Neural Procedural Memory: Empowering LLM Agents with Implicit Activation Steering + +--- +type: paper +title: "Neural Procedural Memory: Empowering LLM Agents with Implicit Activation Steering" +authors: Chengfeng Zhao, Yuqiao Tan, Shizhu He, Yequan Wang, Jun Zhao, Kang Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29824 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 24 +collection_queries: agent-evaluation, agent-memory, autonomous-agent-llm, llm-agent, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, agent-memory, autonomous-agent-llm, llm-agent, rag-agent +- inferred topics: agent-evaluation, computer-use, memory, rag, tool-use, workflow-agent +- arXiv categories: cs.CL, cs.AI +- collection score: 24 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29824 diff --git a/papers/items/2026-2606-29894-saber-math-automated-benchmark-for-information-retrieval-evaluation-in-mathemati.md b/papers/items/2026-2606-29894-saber-math-automated-benchmark-for-information-retrieval-evaluation-in-mathemati.md new file mode 100644 index 0000000..be18556 --- /dev/null +++ b/papers/items/2026-2606-29894-saber-math-automated-benchmark-for-information-retrieval-evaluation-in-mathemati.md @@ -0,0 +1,62 @@ +# Paper: SABER-Math: Automated Benchmark for Information Retrieval Evaluation in Mathematics + +--- +type: paper +title: "SABER-Math: Automated Benchmark for Information Retrieval Evaluation in Mathematics" +authors: Nikolay Georgiev, Maria Drencheva, Kseniia Ibragimova, Ivo Petrov, Dimitar I. Dimitrov, Martin Vechev +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29894 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, rag +- arXiv categories: cs.IR, cs.AI, cs.CL, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29894 diff --git a/papers/items/2026-2606-29914-memdelta-controlled-baselines-and-hidden-confounds-in-agent-memory-evaluation.md b/papers/items/2026-2606-29914-memdelta-controlled-baselines-and-hidden-confounds-in-agent-memory-evaluation.md new file mode 100644 index 0000000..b8eda4c --- /dev/null +++ b/papers/items/2026-2606-29914-memdelta-controlled-baselines-and-hidden-confounds-in-agent-memory-evaluation.md @@ -0,0 +1,61 @@ +# Paper: MemDelta: Controlled Baselines and Hidden Confounds in Agent Memory Evaluation + +--- +type: paper +title: "MemDelta: Controlled Baselines and Hidden Confounds in Agent Memory Evaluation" +authors: Kuan Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29914 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, rag-agent +- inferred topics: agent-evaluation, memory, rag +- arXiv categories: cs.CL, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29914 diff --git a/papers/items/2026-2606-29932-saga-scene-aware-goal-evolving-agents-for-long-horizon-civrealm-strategy-plannin.md b/papers/items/2026-2606-29932-saga-scene-aware-goal-evolving-agents-for-long-horizon-civrealm-strategy-plannin.md new file mode 100644 index 0000000..1427cb2 --- /dev/null +++ b/papers/items/2026-2606-29932-saga-scene-aware-goal-evolving-agents-for-long-horizon-civrealm-strategy-plannin.md @@ -0,0 +1,63 @@ +# Paper: SAGA: Scene-Aware, Goal-Evolving Agents for Long-Horizon CivRealm Strategy Planning + +--- +type: paper +title: "SAGA: Scene-Aware, Goal-Evolving Agents for Long-Horizon CivRealm Strategy Planning" +authors: Tianyu Jin, Shuo Chen, Yida Wang, Liuyu Xiang, Yingzhuo Liu, Zhiyao Jiang, Yexin Li, Zhaofeng He +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29932 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29932 diff --git a/papers/items/2026-2606-29957-swe-together-evaluating-coding-agents-in-interactive-user-sessions.md b/papers/items/2026-2606-29957-swe-together-evaluating-coding-agents-in-interactive-user-sessions.md new file mode 100644 index 0000000..f881d33 --- /dev/null +++ b/papers/items/2026-2606-29957-swe-together-evaluating-coding-agents-in-interactive-user-sessions.md @@ -0,0 +1,61 @@ +# Paper: SWE-Together: Evaluating Coding Agents in Interactive User Sessions + +--- +type: paper +title: "SWE-Together: Evaluating Coding Agents in Interactive User Sessions" +authors: Yifan Wu, Zhuokai Zhao, Songlin Li, Ho Hin Lee, Jiacheng Zhu, Shirley Wu, Tianhe Yu, Serena Li, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29957 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-evaluation, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, coding-agent +- inferred topics: agent-evaluation, coding-agent, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29957 diff --git a/papers/items/2026-2606-29961-duomem-towards-capable-on-device-memory-agents-via-dual-space-distillation.md b/papers/items/2026-2606-29961-duomem-towards-capable-on-device-memory-agents-via-dual-space-distillation.md new file mode 100644 index 0000000..e50f060 --- /dev/null +++ b/papers/items/2026-2606-29961-duomem-towards-capable-on-device-memory-agents-via-dual-space-distillation.md @@ -0,0 +1,61 @@ +# Paper: DuoMem: Towards Capable On-Device Memory Agents via Dual-Space Distillation + +--- +type: paper +title: "DuoMem: Towards Capable On-Device Memory Agents via Dual-Space Distillation" +authors: Peyman Hosseini, Ondrej Bohdal, Ahmed Alajrami, Andrea Maracani, Ignacio Castro, Matthew Purver, Mete Ozay, Savas Ozkan, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.29961 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, embodied-agent, memory +- arXiv categories: cs.LG, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.29961 diff --git a/papers/items/2026-2606-30005-llm-agents-are-latent-context-managers-eliciting-self-managed-context-via-a-prop.md b/papers/items/2026-2606-30005-llm-agents-are-latent-context-managers-eliciting-self-managed-context-via-a-prop.md new file mode 100644 index 0000000..fe76151 --- /dev/null +++ b/papers/items/2026-2606-30005-llm-agents-are-latent-context-managers-eliciting-self-managed-context-via-a-prop.md @@ -0,0 +1,60 @@ +# Paper: LLM Agents Are Latent Context Managers: Eliciting Self-Managed Context via a Proprioceptive Dashboard + +--- +type: paper +title: "LLM Agents Are Latent Context Managers: Eliciting Self-Managed Context via a Proprioceptive Dashboard" +authors: Binyan Xu, Haitao Li, Kehuan Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30005 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - memory + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: memory, planning, tool-use +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30005 diff --git a/papers/items/2026-2606-30111-automating-the-design-of-embodied-agent-architectures.md b/papers/items/2026-2606-30111-automating-the-design-of-embodied-agent-architectures.md new file mode 100644 index 0000000..10b3b39 --- /dev/null +++ b/papers/items/2026-2606-30111-automating-the-design-of-embodied-agent-architectures.md @@ -0,0 +1,66 @@ +# Paper: Automating the Design of Embodied Agent Architectures + +--- +type: paper +title: Automating the Design of Embodied Agent Architectures +authors: Jian Zhou, Sihao Lin, Jin Li, Shuai Fu, Gengze Zhou, Qi Wu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30111 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - embodied-agent + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, embodied-agent, memory, planning, reasoning, tool-use +- arXiv categories: cs.RO, cs.AI, cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30111 diff --git a/papers/items/2026-2606-30119-on-the-internet-nobody-knows-you-re-an-llm-bot-unmasking-web-agents-with-multi-l.md b/papers/items/2026-2606-30119-on-the-internet-nobody-knows-you-re-an-llm-bot-unmasking-web-agents-with-multi-l.md new file mode 100644 index 0000000..cb568f7 --- /dev/null +++ b/papers/items/2026-2606-30119-on-the-internet-nobody-knows-you-re-an-llm-bot-unmasking-web-agents-with-multi-l.md @@ -0,0 +1,63 @@ +# Paper: On the Internet, Nobody Knows You're an LLM Bot: Unmasking Web Agents with Multi-Layer Fingerprinting + +--- +type: paper +title: "On the Internet, Nobody Knows You're an LLM Bot: Unmasking Web Agents with Multi-Layer Fingerprinting" +authors: Iliana Fayolle, Sihem Bouhenniche, Samuel Pélissier, Pierre Laperdrix, Clémentine Maurice, Walter Rudametkin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30119 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - embodied-agent + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, embodied-agent, rag, tool-use, workflow-agent +- arXiv categories: cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30119 diff --git a/papers/items/2026-2606-30185-dynamo-dynamic-skill-tool-evolution-for-vision-language-agents.md b/papers/items/2026-2606-30185-dynamo-dynamic-skill-tool-evolution-for-vision-language-agents.md new file mode 100644 index 0000000..93eac02 --- /dev/null +++ b/papers/items/2026-2606-30185-dynamo-dynamic-skill-tool-evolution-for-vision-language-agents.md @@ -0,0 +1,60 @@ +# Paper: Dynamo: Dynamic Skill-Tool Evolution for Vision-Language Agents + +--- +type: paper +title: "Dynamo: Dynamic Skill-Tool Evolution for Vision-Language Agents" +authors: Yutao Sun, Yanting Miao, Hao-Xuan Ma, Mengyu Zhou, Mingshuai Chen, Tiancheng Zhao, Dexin Wang, Lei Lv, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30185 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30185 diff --git a/papers/items/2026-2606-30251-taco-tool-augmented-credit-optimization-for-agentic-tool-use.md b/papers/items/2026-2606-30251-taco-tool-augmented-credit-optimization-for-agentic-tool-use.md new file mode 100644 index 0000000..418a128 --- /dev/null +++ b/papers/items/2026-2606-30251-taco-tool-augmented-credit-optimization-for-agentic-tool-use.md @@ -0,0 +1,61 @@ +# Paper: TACO: Tool-Augmented Credit Optimization for Agentic Tool Use + +--- +type: paper +title: "TACO: Tool-Augmented Credit Optimization for Agentic Tool Use" +authors: Mingkuan Feng, Jinyang Wu, Hao Gu, Fangrui Lv, Ruihan Jin, Chuyuan Zhang, Zhengqi Wen, Jianhua Tao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30251 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, computer-use, reasoning, tool-use +- arXiv categories: cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30251 diff --git a/papers/items/2026-2606-30259-multi-agentic-system-leveraging-open-source-llms-to-mitigate-disinformation-thre.md b/papers/items/2026-2606-30259-multi-agentic-system-leveraging-open-source-llms-to-mitigate-disinformation-thre.md new file mode 100644 index 0000000..130a4ef --- /dev/null +++ b/papers/items/2026-2606-30259-multi-agentic-system-leveraging-open-source-llms-to-mitigate-disinformation-thre.md @@ -0,0 +1,60 @@ +# Paper: Multi-Agentic System Leveraging Open-Source LLMs to Mitigate Disinformation Threats + +--- +type: paper +title: Multi-Agentic System Leveraging Open-Source LLMs to Mitigate Disinformation Threats +authors: Sebastian Kula, Martin Tamajka +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30259 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, rag +- arXiv categories: cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30259 diff --git a/papers/items/2026-2606-30266-towards-continual-motion-language-agents-lora-variants-for-incremental-motion-un.md b/papers/items/2026-2606-30266-towards-continual-motion-language-agents-lora-variants-for-incremental-motion-un.md new file mode 100644 index 0000000..e6651b8 --- /dev/null +++ b/papers/items/2026-2606-30266-towards-continual-motion-language-agents-lora-variants-for-incremental-motion-un.md @@ -0,0 +1,59 @@ +# Paper: Towards Continual Motion-Language Agents: LoRA Variants for Incremental Motion Understanding and Generation + +--- +type: paper +title: "Towards Continual Motion-Language Agents: LoRA Variants for Incremental Motion Understanding and Generation" +authors: Bertram Taetz, Hugo Albuquerque Cosme da Silva, Gabriele Bleser-Taetz +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30266 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm, language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, language-agent +- inferred topics: agent-evaluation +- arXiv categories: cs.LG, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30266 diff --git a/papers/items/2026-2606-30294-rehearsed-multi-agent-live-product-demonstrations-with-real-time-voice-question-.md b/papers/items/2026-2606-30294-rehearsed-multi-agent-live-product-demonstrations-with-real-time-voice-question-.md new file mode 100644 index 0000000..1a3affc --- /dev/null +++ b/papers/items/2026-2606-30294-rehearsed-multi-agent-live-product-demonstrations-with-real-time-voice-question-.md @@ -0,0 +1,66 @@ +# Paper: Rehearsed Multi-Agent Live Product Demonstrations with Real-Time Voice Question Answering + +--- +type: paper +title: Rehearsed Multi-Agent Live Product Demonstrations with Real-Time Voice Question Answering +authors: Rahul Khedar, Mayank Malhotra, Avinash Karn, Mouli V, Prakhar Mehrotra +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30294 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - multi-agent + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.HC + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, multi-agent, rag, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.HC, cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30294 diff --git a/papers/items/2026-2606-30383-whose-side-is-your-agent-on-multi-party-principal-loyalty-in-llm-agents.md b/papers/items/2026-2606-30383-whose-side-is-your-agent-on-multi-party-principal-loyalty-in-llm-agents.md new file mode 100644 index 0000000..33cb107 --- /dev/null +++ b/papers/items/2026-2606-30383-whose-side-is-your-agent-on-multi-party-principal-loyalty-in-llm-agents.md @@ -0,0 +1,60 @@ +# Paper: Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents + +--- +type: paper +title: Whose Side Is Your Agent On? Multi-Party Principal Loyalty in LLM Agents +authors: Bojie Li, Noah Shi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30383 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30383 diff --git a/papers/items/2026-2606-30454-collective-cooperation-without-individual-fidelity-in-llm-agents.md b/papers/items/2026-2606-30454-collective-cooperation-without-individual-fidelity-in-llm-agents.md new file mode 100644 index 0000000..3a6cee2 --- /dev/null +++ b/papers/items/2026-2606-30454-collective-cooperation-without-individual-fidelity-in-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: Collective cooperation without individual fidelity in LLM agents + +--- +type: paper +title: Collective cooperation without individual fidelity in LLM agents +authors: Henrique Ferraz de Arruda, Carlos Gracia Lázaro, Alberto Aleta, Yamir Moreno +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30454 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - physics.soc-ph + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, tool-use, world-model +- arXiv categories: physics.soc-ph, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30454 diff --git a/papers/items/2026-2606-30524-the-illusion-of-agentic-complexity-in-readme-md-generation-evaluating-single-age.md b/papers/items/2026-2606-30524-the-illusion-of-agentic-complexity-in-readme-md-generation-evaluating-single-age.md new file mode 100644 index 0000000..81ba607 --- /dev/null +++ b/papers/items/2026-2606-30524-the-illusion-of-agentic-complexity-in-readme-md-generation-evaluating-single-age.md @@ -0,0 +1,63 @@ +# Paper: The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems + +--- +type: paper +title: "The Illusion of Agentic Complexity in README.md Generation: Evaluating Single-Agent vs. Multi-Agent RAG Systems" +authors: Abu Saleh, Tesfay Welegebreal Tesfay, Phuong T. Nguyen, Juri Di Rocco, Muhammad Umar Zeshan, Davide Di Ruscio +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30524 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - multi-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: multi-agent-llm, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm, rag-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, multi-agent, planning, rag +- arXiv categories: cs.SE +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30524 diff --git a/papers/items/2026-2606-30546-mas-lab-a-specification-driven-validation-framework-for-reliable-multi-agent-sys.md b/papers/items/2026-2606-30546-mas-lab-a-specification-driven-validation-framework-for-reliable-multi-agent-sys.md new file mode 100644 index 0000000..bb7c8a5 --- /dev/null +++ b/papers/items/2026-2606-30546-mas-lab-a-specification-driven-validation-framework-for-reliable-multi-agent-sys.md @@ -0,0 +1,62 @@ +# Paper: MAS-Lab: A Specification-Driven Validation Framework for Reliable Multi-Agent Systems + +--- +type: paper +title: "MAS-Lab: A Specification-Driven Validation Framework for Reliable Multi-Agent Systems" +authors: Jordan Augé, Giovanna Carofiglio, Giulio Grassi, Jacques Samain +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30546 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, memory, multi-agent, tool-use, workflow-agent +- arXiv categories: cs.MA +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30546 diff --git a/papers/items/2026-2606-30555-linguistic-firewall-geometry-as-defense-in-multi-agent-systems-routing.md b/papers/items/2026-2606-30555-linguistic-firewall-geometry-as-defense-in-multi-agent-systems-routing.md new file mode 100644 index 0000000..e822323 --- /dev/null +++ b/papers/items/2026-2606-30555-linguistic-firewall-geometry-as-defense-in-multi-agent-systems-routing.md @@ -0,0 +1,64 @@ +# Paper: Linguistic Firewall: Geometry as Defense in Multi-Agent Systems Routing + +--- +type: paper +title: "Linguistic Firewall: Geometry as Defense in Multi-Agent Systems Routing" +authors: Dvir Alsheich, Adar Peleg, Ben Hagag, Rom Himelstein, Amit Levi, Avi Mendelson +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30555 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.MA +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30555 diff --git a/papers/items/2026-2606-30560-tracelab-characterizing-coding-agent-workloads-for-llm-serving.md b/papers/items/2026-2606-30560-tracelab-characterizing-coding-agent-workloads-for-llm-serving.md new file mode 100644 index 0000000..d9f1ee0 --- /dev/null +++ b/papers/items/2026-2606-30560-tracelab-characterizing-coding-agent-workloads-for-llm-serving.md @@ -0,0 +1,62 @@ +# Paper: TraceLab: Characterizing Coding Agent Workloads for LLM Serving + +--- +type: paper +title: "TraceLab: Characterizing Coding Agent Workloads for LLM Serving" +authors: Kan Zhu, Mathew Jacob, Chenxi Ma, Yi Pan, Stephanie Wang, Arvind Krishnamurthy, Baris Kasikci +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30560 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.PF +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, tool-use +- arXiv categories: cs.LG, cs.AI, cs.PF +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30560 diff --git a/papers/items/2026-2606-30566-forensic-trajectory-signatures-for-agent-memory-poisoning-detection.md b/papers/items/2026-2606-30566-forensic-trajectory-signatures-for-agent-memory-poisoning-detection.md new file mode 100644 index 0000000..d8e9a61 --- /dev/null +++ b/papers/items/2026-2606-30566-forensic-trajectory-signatures-for-agent-memory-poisoning-detection.md @@ -0,0 +1,63 @@ +# Paper: Forensic Trajectory Signatures for Agent Memory Poisoning Detection + +--- +type: paper +title: Forensic Trajectory Signatures for Agent Memory Poisoning Detection +authors: Jun Wen Leong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30566 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, llm-agent +- inferred topics: agent-evaluation, computer-use, memory, rag, tool-use +- arXiv categories: cs.CR, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30566 diff --git a/papers/items/2026-2606-30573-swe-interact-reimagining-swe-benchmarks-as-user-driven-long-horizon-coding-sessi.md b/papers/items/2026-2606-30573-swe-interact-reimagining-swe-benchmarks-as-user-driven-long-horizon-coding-sessi.md new file mode 100644 index 0000000..73fa75a --- /dev/null +++ b/papers/items/2026-2606-30573-swe-interact-reimagining-swe-benchmarks-as-user-driven-long-horizon-coding-sessi.md @@ -0,0 +1,63 @@ +# Paper: SWE-INTERACT: Reimagining SWE Benchmarks as User-Driven Long-Horizon Coding Sessions + +--- +type: paper +title: "SWE-INTERACT: Reimagining SWE Benchmarks as User-Driven Long-Horizon Coding Sessions" +authors: Mohit Raghavendra, Anisha Gunjal, Aakash Sabharwal, Yunzhong He +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30573 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, planning, tool-use, workflow-agent +- arXiv categories: cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30573 diff --git a/papers/items/2026-2606-30602-mesa-prioritizing-vulnerable-communication-channels-for-securing-multi-agent-sys.md b/papers/items/2026-2606-30602-mesa-prioritizing-vulnerable-communication-channels-for-securing-multi-agent-sys.md new file mode 100644 index 0000000..1d93896 --- /dev/null +++ b/papers/items/2026-2606-30602-mesa-prioritizing-vulnerable-communication-channels-for-securing-multi-agent-sys.md @@ -0,0 +1,62 @@ +# Paper: MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems + +--- +type: paper +title: "MESA: Prioritizing Vulnerable Communication Channels for Securing Multi-Agent Systems" +authors: Kunyang Li, Kyle Domico, Jonathan Gregory, Patrick McDaniel +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30602 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30602 diff --git a/papers/items/2026-2606-30616-scaling-the-horizon-not-the-parameters-reaching-trillion-parameter-performance-w.md b/papers/items/2026-2606-30616-scaling-the-horizon-not-the-parameters-reaching-trillion-parameter-performance-w.md new file mode 100644 index 0000000..3b698a7 --- /dev/null +++ b/papers/items/2026-2606-30616-scaling-the-horizon-not-the-parameters-reaching-trillion-parameter-performance-w.md @@ -0,0 +1,63 @@ +# Paper: Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent + +--- +type: paper +title: "Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent" +authors: Lei Bai, Zongsheng Cao, Yang Chen, Zhiyao Cui, Shangheng Du, Yue Fan, Shiyang Feng, Zijie Guo, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30616 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, planning, rag, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30616 diff --git a/papers/items/2026-2606-30639-self-evolving-world-models-for-llm-agent-planning.md b/papers/items/2026-2606-30639-self-evolving-world-models-for-llm-agent-planning.md new file mode 100644 index 0000000..72cf0d3 --- /dev/null +++ b/papers/items/2026-2606-30639-self-evolving-world-models-for-llm-agent-planning.md @@ -0,0 +1,65 @@ +# Paper: Self-Evolving World Models for LLM Agent Planning + +--- +type: paper +title: Self-Evolving World Models for LLM Agent Planning +authors: Xuan Zhang, Wenxuan Zhang, See-Kiong Ng, Yang Deng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30639 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - reasoning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: llm-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, planning-agent +- inferred topics: agent-evaluation, memory, planning, rag, reasoning, tool-use, world-model +- arXiv categories: cs.AI, cs.CL +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30639 diff --git a/papers/items/2026-2606-30697-lumos-a-semantic-operating-system-layer-for-accessibility-grounded-ai-agents.md b/papers/items/2026-2606-30697-lumos-a-semantic-operating-system-layer-for-accessibility-grounded-ai-agents.md new file mode 100644 index 0000000..87e0f20 --- /dev/null +++ b/papers/items/2026-2606-30697-lumos-a-semantic-operating-system-layer-for-accessibility-grounded-ai-agents.md @@ -0,0 +1,63 @@ +# Paper: LUMOS: A Semantic Operating-System Layer for Accessibility-Grounded AI Agents + +--- +type: paper +title: "LUMOS: A Semantic Operating-System Layer for Accessibility-Grounded AI Agents" +authors: Yogeswar Reddy Thota +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30697 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - computer-use + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.OS + - cs.AI + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: ai-agent, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, web-gui-agent +- inferred topics: computer-use, rag, tool-use, workflow-agent +- arXiv categories: cs.OS, cs.AI, cs.CV +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30697 diff --git a/papers/items/2026-2606-30755-understanding-and-evaluating-claw-like-agent-security-through-a-computer-systems.md b/papers/items/2026-2606-30755-understanding-and-evaluating-claw-like-agent-security-through-a-computer-systems.md new file mode 100644 index 0000000..70b7a53 --- /dev/null +++ b/papers/items/2026-2606-30755-understanding-and-evaluating-claw-like-agent-security-through-a-computer-systems.md @@ -0,0 +1,61 @@ +# Paper: Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens + +--- +type: paper +title: Understanding and Evaluating Claw-like Agent Security Through a Computer-Systems Lens +authors: Peizhi Niu, Wenjie Qu, Shangding Gu, Tianneng Shi, Yuankai Li, Ahmad Tawaha, Hend Alzahrani, Vincent Siu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30755 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety, ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, ai-agent +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30755 diff --git a/papers/items/2026-2606-30840-contrastive-reflection-for-iterative-prompt-optimization.md b/papers/items/2026-2606-30840-contrastive-reflection-for-iterative-prompt-optimization.md new file mode 100644 index 0000000..85ae508 --- /dev/null +++ b/papers/items/2026-2606-30840-contrastive-reflection-for-iterative-prompt-optimization.md @@ -0,0 +1,62 @@ +# Paper: Contrastive Reflection for Iterative Prompt Optimization + +--- +type: paper +title: Contrastive Reflection for Iterative Prompt Optimization +authors: Derek Koh, Jinghui Mo, Benjamin H. Le, Jiening Zhan, Baofen Zheng, Kevin Bevis, Nathaniel C. Owen, Lauren Elizabeth Charney, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30840 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: ai-agent, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, llm-agent +- inferred topics: agent-evaluation, computer-use, rag, reasoning, workflow-agent +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30840 diff --git a/papers/items/2026-2606-30877-a-systematic-approach-to-multi-agent-ai-from-advanced-regulatory-control-theory-.md b/papers/items/2026-2606-30877-a-systematic-approach-to-multi-agent-ai-from-advanced-regulatory-control-theory-.md new file mode 100644 index 0000000..202ac3d --- /dev/null +++ b/papers/items/2026-2606-30877-a-systematic-approach-to-multi-agent-ai-from-advanced-regulatory-control-theory-.md @@ -0,0 +1,62 @@ +# Paper: A Systematic Approach to Multi-Agent AI from Advanced Regulatory Control Theory: Safe and Auditable LLM Operator Agents for Process Control + +--- +type: paper +title: "A Systematic Approach to Multi-Agent AI from Advanced Regulatory Control Theory: Safe and Auditable LLM Operator Agents for Process Control" +authors: Idelfonso B. R. Nogueira, Sigurd Skogestad +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30877 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, tool-use +- arXiv categories: eess.SY, cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30877 diff --git a/papers/items/2026-2606-30887-training-therapeutic-judges-and-multi-agent-systems-for-human-aligned-mental-hea.md b/papers/items/2026-2606-30887-training-therapeutic-judges-and-multi-agent-systems-for-human-aligned-mental-hea.md new file mode 100644 index 0000000..cd20a93 --- /dev/null +++ b/papers/items/2026-2606-30887-training-therapeutic-judges-and-multi-agent-systems-for-human-aligned-mental-hea.md @@ -0,0 +1,63 @@ +# Paper: Training Therapeutic Judges and Multi-Agent Systems for Human-Aligned Mental Health Support + +--- +type: paper +title: Training Therapeutic Judges and Multi-Agent Systems for Human-Aligned Mental Health Support +authors: Mizanur Rahman, Abeer Badawi, Elahe Rahimi, Laleh Seyyed-Kalantari, Frank Rudzicz, Enamul Hoque, Elham Dolatabadi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30887 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, tool-use +- arXiv categories: cs.CL, cs.AI, cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30887 diff --git a/papers/items/2026-2606-30906-investigating-multi-agent-deliberation-in-law.md b/papers/items/2026-2606-30906-investigating-multi-agent-deliberation-in-law.md new file mode 100644 index 0000000..9bd1e5b --- /dev/null +++ b/papers/items/2026-2606-30906-investigating-multi-agent-deliberation-in-law.md @@ -0,0 +1,61 @@ +# Paper: Investigating Multi-Agent Deliberation in Law + +--- +type: paper +title: Investigating Multi-Agent Deliberation in Law +authors: Cor Steging, Ludi van Leeuwen, Tadeusz Zbiegień +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30906 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agentic-ai, ai-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, ai-agent, multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30906 diff --git a/papers/items/2026-2606-30949-agrefactor-self-evolving-agentic-workflow-for-hls-compatibility-and-performance.md b/papers/items/2026-2606-30949-agrefactor-self-evolving-agentic-workflow-for-hls-compatibility-and-performance.md new file mode 100644 index 0000000..3a2c28b --- /dev/null +++ b/papers/items/2026-2606-30949-agrefactor-self-evolving-agentic-workflow-for-hls-compatibility-and-performance.md @@ -0,0 +1,64 @@ +# Paper: AgRefactor: Self-Evolving Agentic Workflow for HLS Compatibility and Performance + +--- +type: paper +title: "AgRefactor: Self-Evolving Agentic Workflow for HLS Compatibility and Performance" +authors: Yang Zou, Zijian Ding, Yizhou Sun, Jason Cong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30949 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.AR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agentic-ai, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, multi-agent-llm +- inferred topics: agent-evaluation, memory, multi-agent, rag, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.AR +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30949 diff --git a/papers/items/2026-2606-30970-behavioral-governance-for-autonomous-ai-agents-the-agentbound-framework.md b/papers/items/2026-2606-30970-behavioral-governance-for-autonomous-ai-agents-the-agentbound-framework.md new file mode 100644 index 0000000..e009f87 --- /dev/null +++ b/papers/items/2026-2606-30970-behavioral-governance-for-autonomous-ai-agents-the-agentbound-framework.md @@ -0,0 +1,61 @@ +# Paper: Behavioral Governance for Autonomous AI Agents: The AgentBound Framework + +--- +type: paper +title: "Behavioral Governance for Autonomous AI Agents: The AgentBound Framework" +authors: Anuj Kaul, Qianlong Lan, Pranay Gupta +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30970 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30970 diff --git a/papers/items/2026-2606-30986-the-organizational-behavior-of-agentic-ai-collective-intelligence-in-human-agent.md b/papers/items/2026-2606-30986-the-organizational-behavior-of-agentic-ai-collective-intelligence-in-human-agent.md new file mode 100644 index 0000000..72503bf --- /dev/null +++ b/papers/items/2026-2606-30986-the-organizational-behavior-of-agentic-ai-collective-intelligence-in-human-agent.md @@ -0,0 +1,65 @@ +# Paper: The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows + +--- +type: paper +title: "The Organizational Behavior of Agentic AI: Collective Intelligence in Human-Agent Workflows" +authors: Canhui Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.30986 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - memory + - planning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY + - cs.HC + - cs.MA + - econ.GN +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agentic-ai, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, llm-agent +- inferred topics: memory, planning, tool-use, workflow-agent, world-model +- arXiv categories: cs.CY, cs.HC, cs.MA, econ.GN +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.30986 diff --git a/papers/items/2026-2606-31046-openlife-toward-open-world-artificial-life-with-autonomous-llm-agents.md b/papers/items/2026-2606-31046-openlife-toward-open-world-artificial-life-with-autonomous-llm-agents.md new file mode 100644 index 0000000..7d427a8 --- /dev/null +++ b/papers/items/2026-2606-31046-openlife-toward-open-world-artificial-life-with-autonomous-llm-agents.md @@ -0,0 +1,60 @@ +# Paper: OpenLife: Toward Open-World Artificial Life with Autonomous LLM Agents + +--- +type: paper +title: "OpenLife: Toward Open-World Artificial Life with Autonomous LLM Agents" +authors: Atsushi Masumori, Itsuki Doi, Norihiro Maruyama, Ryosuke Takata, Takashi Ikegami +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31046 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, tool-use +- inferred topics: agent-evaluation, memory, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31046 diff --git a/papers/items/2026-2606-31073-multiuav-plat-an-llm-oriented-platform-benchmark-and-framework-for-multi-uav-col.md b/papers/items/2026-2606-31073-multiuav-plat-an-llm-oriented-platform-benchmark-and-framework-for-multi-uav-col.md new file mode 100644 index 0000000..6038ccc --- /dev/null +++ b/papers/items/2026-2606-31073-multiuav-plat-an-llm-oriented-platform-benchmark-and-framework-for-multi-uav-col.md @@ -0,0 +1,67 @@ +# Paper: MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning + +--- +type: paper +title: "MultiUAV-Plat: An LLM-Oriented Platform, Benchmark and Framework for Multi-UAV Collaborative Task Planning" +authors: Sheng Zhang, Qinglin Li, Yuechao Zang, Xueqin Huang, Yijia Fu, Cheng Zhu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31073 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - multi-agent + - planning + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-evaluation, llm-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, llm-agent, planning-agent +- inferred topics: agent-evaluation, embodied-agent, memory, multi-agent, planning, rag, tool-use, world-model +- arXiv categories: cs.AI, cs.MA, cs.RO +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31073 diff --git a/papers/items/2026-2606-31085-ddiagents-mechanism-conditioned-context-flow-for-drug-drug-interaction-predictio.md b/papers/items/2026-2606-31085-ddiagents-mechanism-conditioned-context-flow-for-drug-drug-interaction-predictio.md new file mode 100644 index 0000000..292cb6d --- /dev/null +++ b/papers/items/2026-2606-31085-ddiagents-mechanism-conditioned-context-flow-for-drug-drug-interaction-predictio.md @@ -0,0 +1,63 @@ +# Paper: DDIAgents: Mechanism-Conditioned Context Flow for Drug-Drug Interaction Prediction + +--- +type: paper +title: "DDIAgents: Mechanism-Conditioned Context Flow for Drug-Drug Interaction Prediction" +authors: Zhenqian Shen, Yu Liu, Xiaoyi Fu, Quanming Yao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31085 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31085 diff --git a/papers/items/2026-2606-31134-beyond-the-library-an-agentic-framework-for-autoformalizing-research-mathematics.md b/papers/items/2026-2606-31134-beyond-the-library-an-agentic-framework-for-autoformalizing-research-mathematics.md new file mode 100644 index 0000000..8f11d4a --- /dev/null +++ b/papers/items/2026-2606-31134-beyond-the-library-an-agentic-framework-for-autoformalizing-research-mathematics.md @@ -0,0 +1,62 @@ +# Paper: Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics + +--- +type: paper +title: "Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics" +authors: Arshia Soltani Moakhar, Iman Gholami, Max Springer, Mahdi JafariRaviz, MohammadTaghi Hajiaghayi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31134 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent, rag, reasoning +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31134 diff --git a/papers/items/2026-2606-31174-clawarena-team-benchmarking-subagent-orchestration-and-dynamic-workflows-in-lang.md b/papers/items/2026-2606-31174-clawarena-team-benchmarking-subagent-orchestration-and-dynamic-workflows-in-lang.md new file mode 100644 index 0000000..1be872e --- /dev/null +++ b/papers/items/2026-2606-31174-clawarena-team-benchmarking-subagent-orchestration-and-dynamic-workflows-in-lang.md @@ -0,0 +1,61 @@ +# Paper: ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents + +--- +type: paper +title: "ClawArena-Team: Benchmarking Subagent Orchestration and Dynamic Workflows in Language-Model Agents" +authors: Kaiwen Xiong, Haonian Ji, Shi Qiu, Zeyu Zheng, Cihang Xie, Xinyu Ye, Huaxiu Yao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31174 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31174 diff --git a/papers/items/2026-2606-31179-healthagentbench-a-unified-benchmark-suite-of-realistic-agentic-healthcare-envir.md b/papers/items/2026-2606-31179-healthagentbench-a-unified-benchmark-suite-of-realistic-agentic-healthcare-envir.md new file mode 100644 index 0000000..c8bdbc9 --- /dev/null +++ b/papers/items/2026-2606-31179-healthagentbench-a-unified-benchmark-suite-of-realistic-agentic-healthcare-envir.md @@ -0,0 +1,63 @@ +# Paper: HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents + +--- +type: paper +title: "HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents" +authors: Qianchu Liu, Sheng Zhang, Guanghui Qin, Jeya Maria Jose Valanarasu, Maximilian Rokuss, Mingyu Lu, Timothy Ossowski, Juan Manuel Zambrano Chaves, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31179 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-evaluation, ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, ai-agent +- inferred topics: agent-evaluation, planning, reasoning, workflow-agent +- arXiv categories: cs.AI, cs.CL, cs.CV +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31179 diff --git a/papers/items/2026-2606-31200-agentic-rag-vlm-affordance-aware-retrieval-augmented-generation-with-self-reflec.md b/papers/items/2026-2606-31200-agentic-rag-vlm-affordance-aware-retrieval-augmented-generation-with-self-reflec.md new file mode 100644 index 0000000..ecad6f7 --- /dev/null +++ b/papers/items/2026-2606-31200-agentic-rag-vlm-affordance-aware-retrieval-augmented-generation-with-self-reflec.md @@ -0,0 +1,62 @@ +# Paper: Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping + +--- +type: paper +title: "Agentic RAG-VLM: Affordance-Aware Retrieval-Augmented Generation with Self-Reflective Planning for Robotic Grasping" +authors: Tao Chen, Lizheng Liu, Jiaxu Wang, Ziyue Jiang, Ruiqi Tian, JiGuang Huo, Zhongxue Gan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31200 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, embodied-agent, planning, rag, reasoning +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31200 diff --git a/papers/items/2026-2606-31209-long-term-traffic-simulation-via-structured-autoregressive-modeling.md b/papers/items/2026-2606-31209-long-term-traffic-simulation-via-structured-autoregressive-modeling.md new file mode 100644 index 0000000..07dec59 --- /dev/null +++ b/papers/items/2026-2606-31209-long-term-traffic-simulation-via-structured-autoregressive-modeling.md @@ -0,0 +1,66 @@ +# Paper: Long-term Traffic Simulation via Structured Autoregressive Modeling + +--- +type: paper +title: Long-term Traffic Simulation via Structured Autoregressive Modeling +authors: Lingyu Xiao, Zexin Feng, Xintao Yan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31209 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - multi-agent + - planning + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, memory, multi-agent, planning, rag, tool-use, world-model +- arXiv categories: cs.AI, cs.RO +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31209 diff --git a/papers/items/2026-2606-31227-securing-the-ai-agent-a-unified-framework-for-multi-layer-agent-red-teaming.md b/papers/items/2026-2606-31227-securing-the-ai-agent-a-unified-framework-for-multi-layer-agent-red-teaming.md new file mode 100644 index 0000000..52574b5 --- /dev/null +++ b/papers/items/2026-2606-31227-securing-the-ai-agent-a-unified-framework-for-multi-layer-agent-red-teaming.md @@ -0,0 +1,59 @@ +# Paper: Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming + +--- +type: paper +title: "Securing the AI Agent: A Unified Framework for Multi-Layer Agent Red Teaming" +authors: Yong Yang, Xing Zheng, Huiyu Wu, Huangsheng Cheng, Xiaorong Shi, Jing Guo, Bo Yang, Yi Zhou, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31227 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-safety, ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, ai-agent +- inferred topics: agent-safety, tool-use +- arXiv categories: cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31227 diff --git a/papers/items/2026-2606-31229-agentic-ideation-sample-efficient-agentic-trajectories-synthesis-for-scientific-.md b/papers/items/2026-2606-31229-agentic-ideation-sample-efficient-agentic-trajectories-synthesis-for-scientific-.md new file mode 100644 index 0000000..35409bd --- /dev/null +++ b/papers/items/2026-2606-31229-agentic-ideation-sample-efficient-agentic-trajectories-synthesis-for-scientific-.md @@ -0,0 +1,63 @@ +# Paper: Agentic-Ideation: Sample Efficient Agentic Trajectories Synthesis for Scientific Ideation Agents + +--- +type: paper +title: "Agentic-Ideation: Sample Efficient Agentic Trajectories Synthesis for Scientific Ideation Agents" +authors: Keyu Zhao, Lingyan Kong, Fengli Xu, Yong Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31229 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - computer-use + - multi-agent + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agentic-ai, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, multi-agent-llm +- inferred topics: computer-use, multi-agent, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31229 diff --git a/papers/items/2026-2606-31252-embodied-cad-solver-grounded-llm-agents-for-parametric-b-rep-assembly-modeling.md b/papers/items/2026-2606-31252-embodied-cad-solver-grounded-llm-agents-for-parametric-b-rep-assembly-modeling.md new file mode 100644 index 0000000..d6829d1 --- /dev/null +++ b/papers/items/2026-2606-31252-embodied-cad-solver-grounded-llm-agents-for-parametric-b-rep-assembly-modeling.md @@ -0,0 +1,62 @@ +# Paper: Embodied CAD: Solver-Grounded LLM Agents for Parametric B-Rep Assembly Modeling + +--- +type: paper +title: "Embodied CAD: Solver-Grounded LLM Agents for Parametric B-Rep Assembly Modeling" +authors: Fumin Liu, Haoyu Zhou, Fei Hao, Lin Yang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31252 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, planning-agent +- inferred topics: agent-evaluation, embodied-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31252 diff --git a/papers/items/2026-2606-31314-a-novel-method-for-differential-algebraic-dynamic-model-discovery-in-power-syste.md b/papers/items/2026-2606-31314-a-novel-method-for-differential-algebraic-dynamic-model-discovery-in-power-syste.md new file mode 100644 index 0000000..475fbd2 --- /dev/null +++ b/papers/items/2026-2606-31314-a-novel-method-for-differential-algebraic-dynamic-model-discovery-in-power-syste.md @@ -0,0 +1,60 @@ +# Paper: A Novel Method for Differential-Algebraic Dynamic Model Discovery in Power Systems: An LLM-Based Multi-Agent Collaborative Framework + +--- +type: paper +title: "A Novel Method for Differential-Algebraic Dynamic Model Discovery in Power Systems: An LLM-Based Multi-Agent Collaborative Framework" +authors: Xinming Wang, Fan Tang, Yingli Wei, Yakun He, Zhe Liu, Ping Jiang, Haoyu Wu, Zihan Guo, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31314 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, computer-use, multi-agent +- arXiv categories: eess.SY +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31314 diff --git a/papers/items/2026-2606-31339-verification-gated-agentic-mission-state-governance-for-intelligent-industrial-m.md b/papers/items/2026-2606-31339-verification-gated-agentic-mission-state-governance-for-intelligent-industrial-m.md new file mode 100644 index 0000000..4d58e2c --- /dev/null +++ b/papers/items/2026-2606-31339-verification-gated-agentic-mission-state-governance-for-intelligent-industrial-m.md @@ -0,0 +1,64 @@ +# Paper: Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems + +--- +type: paper +title: Verification-Gated Agentic Mission-State Governance for Intelligent Industrial Multi-Robot Systems +authors: Guoqin Tang, Qingxuan Jia, Yichen Tan, Zeyuan Huang, Ning Ji, Gang Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31339 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - embodied-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, agent-safety, embodied-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.RO +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31339 diff --git a/papers/items/2026-2606-31410-xiaomi-gui-0-technical-report.md b/papers/items/2026-2606-31410-xiaomi-gui-0-technical-report.md new file mode 100644 index 0000000..122a286 --- /dev/null +++ b/papers/items/2026-2606-31410-xiaomi-gui-0-technical-report.md @@ -0,0 +1,65 @@ +# Paper: Xiaomi-GUI-0 Technical Report + +--- +type: paper +title: Xiaomi-GUI-0 Technical Report +authors: Wanxia Cao, Chengzhen Duan, Pei Fu, Pengzhi Gao, Niu Lian, Fazhan Liu, Hui Liu, Heng Qu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31410 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - embodied-agent + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, embodied-agent, memory, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31410 diff --git a/papers/items/2026-2606-31471-think-while-you-map-asynchronous-vision-language-agents-for-incremental-3d-scene.md b/papers/items/2026-2606-31471-think-while-you-map-asynchronous-vision-language-agents-for-incremental-3d-scene.md new file mode 100644 index 0000000..36838b9 --- /dev/null +++ b/papers/items/2026-2606-31471-think-while-you-map-asynchronous-vision-language-agents-for-incremental-3d-scene.md @@ -0,0 +1,59 @@ +# Paper: Think While You Map: Asynchronous Vision-Language Agents for Incremental 3D Scene Graphs + +--- +type: paper +title: "Think While You Map: Asynchronous Vision-Language Agents for Incremental 3D Scene Graphs" +authors: Deniz Bickici, Michael Pabst, Shohei Mori, Dieter Schmalstieg +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31471 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, rag +- arXiv categories: cs.CV +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31471 diff --git a/papers/items/2026-2606-31612-what-memory-do-gui-agents-really-need-from-passive-records-to-active-task-drivin.md b/papers/items/2026-2606-31612-what-memory-do-gui-agents-really-need-from-passive-records-to-active-task-drivin.md new file mode 100644 index 0000000..e66f00e --- /dev/null +++ b/papers/items/2026-2606-31612-what-memory-do-gui-agents-really-need-from-passive-records-to-active-task-drivin.md @@ -0,0 +1,64 @@ +# Paper: What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States + +--- +type: paper +title: What Memory Do GUI Agents Really Need? From Passive Records to Active Task-Driving States +authors: Chen Liu, Ling Chen, Hanzhang Zhou, Xu Zhang, Quyu Kong, Panrong Tong, Wenhao Wang, Xin Yu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31612 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-memory, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, web-gui-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.CV +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31612 diff --git a/papers/items/2026-2606-31635-a-tutorial-on-autonomous-fault-tolerant-control-using-knowledge-grounded-llm-age.md b/papers/items/2026-2606-31635-a-tutorial-on-autonomous-fault-tolerant-control-using-knowledge-grounded-llm-age.md new file mode 100644 index 0000000..06fc564 --- /dev/null +++ b/papers/items/2026-2606-31635-a-tutorial-on-autonomous-fault-tolerant-control-using-knowledge-grounded-llm-age.md @@ -0,0 +1,63 @@ +# Paper: A Tutorial on Autonomous Fault-Tolerant Control Using Knowledge-Grounded LLM Agents + +--- +type: paper +title: A Tutorial on Autonomous Fault-Tolerant Control Using Knowledge-Grounded LLM Agents +authors: Javal Vyas, Milapji Singh Gill, Artan Markaj, Felix Gehlhoff, Mehmet Mercangöz +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31635 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-safety + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-safety, planning, tool-use, world-model +- arXiv categories: eess.SY, cs.AI, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31635 diff --git a/papers/items/2026-2606-31639-a-lifecycle-and-application-stack-survey-of-large-language-model-vulnerabilities.md b/papers/items/2026-2606-31639-a-lifecycle-and-application-stack-survey-of-large-language-model-vulnerabilities.md new file mode 100644 index 0000000..3d075d3 --- /dev/null +++ b/papers/items/2026-2606-31639-a-lifecycle-and-application-stack-survey-of-large-language-model-vulnerabilities.md @@ -0,0 +1,69 @@ +# Paper: A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems + +--- +type: paper +title: "A Lifecycle and Application-Stack Survey of Large Language Model Vulnerabilities: Attacks, Risks, Defenses, and Open Problems" +authors: Seyed Bagher Hashemi Natanzi, Bo Tang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31639 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - embodied-agent + - memory + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.GT + - cs.LO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation, autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, coding-agent, embodied-agent, memory, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI, cs.GT, cs.LO +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31639 diff --git a/papers/items/2026-2606-31648-think-in-english-answer-in-korean-efficient-adaptation-of-multilingual-tool-usin.md b/papers/items/2026-2606-31648-think-in-english-answer-in-korean-efficient-adaptation-of-multilingual-tool-usin.md new file mode 100644 index 0000000..afc4a76 --- /dev/null +++ b/papers/items/2026-2606-31648-think-in-english-answer-in-korean-efficient-adaptation-of-multilingual-tool-usin.md @@ -0,0 +1,63 @@ +# Paper: Think in English, Answer in Korean: Efficient Adaptation of Multilingual Tool-Using Agents + +--- +type: paper +title: "Think in English, Answer in Korean: Efficient Adaptation of Multilingual Tool-Using Agents" +authors: Utsav Garg, Sungjin Hong, Jason Jung, Justin Lee, Shaan Desai, Joon Hee Kim, Anirudh Shrinivason, Edmond Wen, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31648 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - memory + - multi-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai, function-calling, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, function-calling, tool-use +- inferred topics: memory, multi-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31648 diff --git a/papers/items/2026-2606-31650-echo-prune-to-act-trace-to-learn-with-selective-turn-memory-in-agentic-rl.md b/papers/items/2026-2606-31650-echo-prune-to-act-trace-to-learn-with-selective-turn-memory-in-agentic-rl.md new file mode 100644 index 0000000..ddf1173 --- /dev/null +++ b/papers/items/2026-2606-31650-echo-prune-to-act-trace-to-learn-with-selective-turn-memory-in-agentic-rl.md @@ -0,0 +1,62 @@ +# Paper: ECHO: Prune to act, trace to learn with selective turn memory in agentic RL + +--- +type: paper +title: "ECHO: Prune to act, trace to learn with selective turn memory in agentic RL" +authors: Zijun Xie, Binbin Zheng, Enlei Gong, Jihua Liu, Yuyang You, Lingfeng Liu, Jiayao Tang, Guanqun Zhao, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31650 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, planning, tool-use +- arXiv categories: cs.LG, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31650 diff --git a/papers/items/2026-2606-31665-forecastagentsearch-towards-a-multi-expert-agent-search-system-for-geopolitical-.md b/papers/items/2026-2606-31665-forecastagentsearch-towards-a-multi-expert-agent-search-system-for-geopolitical-.md new file mode 100644 index 0000000..0368d15 --- /dev/null +++ b/papers/items/2026-2606-31665-forecastagentsearch-towards-a-multi-expert-agent-search-system-for-geopolitical-.md @@ -0,0 +1,61 @@ +# Paper: ForecastAgentSearch: Towards a Multi-Expert Agent Search System for Geopolitical Event Forecasting + +--- +type: paper +title: "ForecastAgentSearch: Towards a Multi-Expert Agent Search System for Geopolitical Event Forecasting" +authors: Miaomiao Cai, He Chang, Yunshan Ma, See-kiong Ng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31665 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, planning, rag +- arXiv categories: cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31665 diff --git a/papers/items/2026-2606-31693-shopx-a-foundation-model-for-intent-to-item-fulfillment-in-agentic-shopping.md b/papers/items/2026-2606-31693-shopx-a-foundation-model-for-intent-to-item-fulfillment-in-agentic-shopping.md new file mode 100644 index 0000000..33e48d1 --- /dev/null +++ b/papers/items/2026-2606-31693-shopx-a-foundation-model-for-intent-to-item-fulfillment-in-agentic-shopping.md @@ -0,0 +1,64 @@ +# Paper: ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping + +--- +type: paper +title: "ShopX: A Foundation Model for Intent-to-Item Fulfillment in Agentic Shopping" +authors: Jiacheng Chen, Tao Zhang, Manxi Lin, Dunxian Huang, Teng Shi, Honghao Fu, Mengyan Li, Xinming Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31693 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, planning-agent +- inferred topics: agent-evaluation, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.IR, cs.AI, cs.CL +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31693 diff --git a/papers/items/2026-2606-31744-a-conversational-agentic-interface-to-physics-based-household-digital-twins-for-.md b/papers/items/2026-2606-31744-a-conversational-agentic-interface-to-physics-based-household-digital-twins-for-.md new file mode 100644 index 0000000..13be2b8 --- /dev/null +++ b/papers/items/2026-2606-31744-a-conversational-agentic-interface-to-physics-based-household-digital-twins-for-.md @@ -0,0 +1,62 @@ +# Paper: A Conversational Agentic Interface to Physics-Based Household Digital Twins for Residential Energy Decision Support + +--- +type: paper +title: A Conversational Agentic Interface to Physics-Based Household Digital Twins for Residential Energy Decision Support +authors: Costas Mylonas, Titos Georgoulakis, Magda Foti +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31744 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, planning, rag, tool-use, world-model +- arXiv categories: eess.SY +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31744 diff --git a/papers/items/2026-2606-31767-jeto-bench-a-reproducible-benchmark-for-execution-time-improvement-patches-in-ja.md b/papers/items/2026-2606-31767-jeto-bench-a-reproducible-benchmark-for-execution-time-improvement-patches-in-ja.md new file mode 100644 index 0000000..1025704 --- /dev/null +++ b/papers/items/2026-2606-31767-jeto-bench-a-reproducible-benchmark-for-execution-time-improvement-patches-in-ja.md @@ -0,0 +1,60 @@ +# Paper: JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java + +--- +type: paper +title: "JETO-Bench: A Reproducible Benchmark for Execution Time Improvement Patches in Java" +authors: Khashayar Etemadi, Zhendong Su +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31767 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, tool-use +- arXiv categories: cs.SE +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31767 diff --git a/papers/items/2026-2606-31831-an-agentic-ai-framework-to-accelerate-scientific-discovery-in-plant-phenotyping.md b/papers/items/2026-2606-31831-an-agentic-ai-framework-to-accelerate-scientific-discovery-in-plant-phenotyping.md new file mode 100644 index 0000000..2ea71f5 --- /dev/null +++ b/papers/items/2026-2606-31831-an-agentic-ai-framework-to-accelerate-scientific-discovery-in-plant-phenotyping.md @@ -0,0 +1,60 @@ +# Paper: An Agentic AI Framework to Accelerate Scientific Discovery in Plant Phenotyping + +--- +type: paper +title: An Agentic AI Framework to Accelerate Scientific Discovery in Plant Phenotyping +authors: Renan Souza, Daniel Rosendo, Kelsey Carter, John Lagergren, Frédéric Suter, Shelaine L. Curd, Gerald A. Tuskan, Rafael Ferreira da Silva, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31831 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-safety + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agentic-ai, ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, ai-agent +- inferred topics: agent-safety, planning, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31831 diff --git a/papers/items/2026-2606-31916-theory-of-mind-and-persuasion-beyond-conversation-assessing-the-capacity-of-llms.md b/papers/items/2026-2606-31916-theory-of-mind-and-persuasion-beyond-conversation-assessing-the-capacity-of-llms.md new file mode 100644 index 0000000..302ea84 --- /dev/null +++ b/papers/items/2026-2606-31916-theory-of-mind-and-persuasion-beyond-conversation-assessing-the-capacity-of-llms.md @@ -0,0 +1,62 @@ +# Paper: Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action + +--- +type: paper +title: "Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action" +authors: Ben Slater, Matteo G. Mecattaf, Lucy G. Cheke, John Burden, Winnie Street +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31916 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, agent-safety, planning, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31916 diff --git a/papers/items/2026-2606-31980-digitalcoach-communication-and-grounding-gaps-in-human-and-agentic-computer-use-.md b/papers/items/2026-2606-31980-digitalcoach-communication-and-grounding-gaps-in-human-and-agentic-computer-use-.md new file mode 100644 index 0000000..895fe16 --- /dev/null +++ b/papers/items/2026-2606-31980-digitalcoach-communication-and-grounding-gaps-in-human-and-agentic-computer-use-.md @@ -0,0 +1,63 @@ +# Paper: DigitalCoach: Communication and Grounding Gaps in Human and Agentic Computer Use Coaching + +--- +type: paper +title: "DigitalCoach: Communication and Grounding Gaps in Human and Agentic Computer Use Coaching" +authors: Meng Chen, Anya Ji, Tsung-Han Wu, Tobias Maringgele, David M. Chan, Alane Suhr, Amy Pavel +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.31980 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, planning, rag +- arXiv categories: cs.CL, cs.AI, cs.HC +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.31980 diff --git a/papers/items/2026-2606-32025-generative-skill-composition-for-llm-agents.md b/papers/items/2026-2606-32025-generative-skill-composition-for-llm-agents.md new file mode 100644 index 0000000..44aadc9 --- /dev/null +++ b/papers/items/2026-2606-32025-generative-skill-composition-for-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: Generative Skill Composition for LLM Agents + +--- +type: paper +title: Generative Skill Composition for LLM Agents +authors: Xinyu Zhao, Zhen Tan, Vaishnav Tadiparthi, Nakul Agarwal, Kwonjoon Lee, Ehsan Moradi Pari, Hossein Nourkhiz Mahjoub, Tianlong Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.32025 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: coding-agent, llm-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, llm-agent, planning-agent +- inferred topics: agent-evaluation, coding-agent, planning, rag, reasoning +- arXiv categories: cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.32025 diff --git a/papers/items/2026-2606-32034-qval-cheaply-evaluating-dense-supervision-signals-for-long-horizon-llm-agents.md b/papers/items/2026-2606-32034-qval-cheaply-evaluating-dense-supervision-signals-for-long-horizon-llm-agents.md new file mode 100644 index 0000000..568051c --- /dev/null +++ b/papers/items/2026-2606-32034-qval-cheaply-evaluating-dense-supervision-signals-for-long-horizon-llm-agents.md @@ -0,0 +1,63 @@ +# Paper: QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents + +--- +type: paper +title: "QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents" +authors: Sergio Hernández-Gutiérrez, Matteo Merler, Ilze Amanda Auzina, Joschka Strüber, Ameya Prabhu, Matthias Bethge +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2606.32034 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2606.32034 diff --git a/papers/items/2026-2607-00038-stop-hand-holding-your-coding-agent-engineering-the-loops-that-replace-step-by-s.md b/papers/items/2026-2607-00038-stop-hand-holding-your-coding-agent-engineering-the-loops-that-replace-step-by-s.md new file mode 100644 index 0000000..ca2d9a8 --- /dev/null +++ b/papers/items/2026-2607-00038-stop-hand-holding-your-coding-agent-engineering-the-loops-that-replace-step-by-s.md @@ -0,0 +1,63 @@ +# Paper: Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting + +--- +type: paper +title: "Stop Hand-Holding Your Coding Agent: Engineering the Loops that Replace Step-by-Step Prompting" +authors: Sandeco Macedo +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00038 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-28 +updated_at: 2026-06-28 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-safety, coding-agent, computer-use, memory, rag, tool-use +- arXiv categories: cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00038 diff --git a/papers/items/2026-2607-00041-atm-cid-brokered-pre-write-admission-for-multi-agent-code-co-synthesis.md b/papers/items/2026-2607-00041-atm-cid-brokered-pre-write-admission-for-multi-agent-code-co-synthesis.md new file mode 100644 index 0000000..966d2b4 --- /dev/null +++ b/papers/items/2026-2607-00041-atm-cid-brokered-pre-write-admission-for-multi-agent-code-co-synthesis.md @@ -0,0 +1,64 @@ +# Paper: ATM: CID-Brokered Pre-Write Admission for Multi-Agent Code Co-Synthesis + +--- +type: paper +title: "ATM: CID-Brokered Pre-Write Admission for Multi-Agent Code Co-Synthesis" +authors: Eagl Huang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00041 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-29 +updated_at: 2026-06-29 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - multi-agent + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, computer-use, multi-agent, planning, rag +- arXiv categories: cs.SE, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00041 diff --git a/papers/items/2026-2607-00233-from-signals-to-structure-how-memory-architecture-drives-language-emergence-in-l.md b/papers/items/2026-2607-00233-from-signals-to-structure-how-memory-architecture-drives-language-emergence-in-l.md new file mode 100644 index 0000000..99064d3 --- /dev/null +++ b/papers/items/2026-2607-00233-from-signals-to-structure-how-memory-architecture-drives-language-emergence-in-l.md @@ -0,0 +1,63 @@ +# Paper: From Signals to Structure: How Memory Architecture Drives Language Emergence in LLM Agents + +--- +type: paper +title: "From Signals to Structure: How Memory Architecture Drives Language Emergence in LLM Agents" +authors: Yashar Talebirad, Eden Redman, Ali Parsaee, Osmar R. Zaiane +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00233 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.IT + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: memory, rag, tool-use +- arXiv categories: cs.AI, cs.CL, cs.IT, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00233 diff --git a/papers/items/2026-2607-00255-slm-llm-or-agentic-ai-toward-intelligent-uav-enabled-wpt-systems-in-low-altitude.md b/papers/items/2026-2607-00255-slm-llm-or-agentic-ai-toward-intelligent-uav-enabled-wpt-systems-in-low-altitude.md new file mode 100644 index 0000000..d08ab8f --- /dev/null +++ b/papers/items/2026-2607-00255-slm-llm-or-agentic-ai-toward-intelligent-uav-enabled-wpt-systems-in-low-altitude.md @@ -0,0 +1,60 @@ +# Paper: SLM, LLM or Agentic AI? Toward Intelligent UAV-Enabled WPT Systems in Low-Altitude Economy Networks + +--- +type: paper +title: SLM, LLM or Agentic AI? Toward Intelligent UAV-Enabled WPT Systems in Low-Altitude Economy Networks +authors: Feibo Jiang, Li Dong, Lei Mao, Kezhi Wang, Xianbin Wang, Abbas Jamalipour +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00255 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - reasoning + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IT +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, reasoning, world-model +- arXiv categories: cs.IT +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00255 diff --git a/papers/items/2026-2607-00297-epc-a-standardized-protocol-for-measuring-evaluator-preference-dynamics-in-llm-a.md b/papers/items/2026-2607-00297-epc-a-standardized-protocol-for-measuring-evaluator-preference-dynamics-in-llm-a.md new file mode 100644 index 0000000..3293abb --- /dev/null +++ b/papers/items/2026-2607-00297-epc-a-standardized-protocol-for-measuring-evaluator-preference-dynamics-in-llm-a.md @@ -0,0 +1,62 @@ +# Paper: EPC: A Standardized Protocol for Measuring Evaluator Preference Dynamics in LLM Agent Systems + +--- +type: paper +title: "EPC: A Standardized Protocol for Measuring Evaluator Preference Dynamics in LLM Agent Systems" +authors: Zewen Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00297 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, tool-use +- arXiv categories: cs.LG, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00297 diff --git a/papers/items/2026-2607-00334-managed-autonomy-at-runtime-gear-based-safety-and-governance-for-single-and-mult.md b/papers/items/2026-2607-00334-managed-autonomy-at-runtime-gear-based-safety-and-governance-for-single-and-mult.md new file mode 100644 index 0000000..ea0f4a8 --- /dev/null +++ b/papers/items/2026-2607-00334-managed-autonomy-at-runtime-gear-based-safety-and-governance-for-single-and-mult.md @@ -0,0 +1,63 @@ +# Paper: Managed Autonomy at Runtime: Gear-Based Safety and Governance for Single- and Multi-Agent Cyber-Physical Systems + +--- +type: paper +title: "Managed Autonomy at Runtime: Gear-Based Safety and Governance for Single- and Multi-Agent Cyber-Physical Systems" +authors: Srini Ramaswamy, Wang Miaosheng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00334 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - embodied-agent + - multi-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: autonomous-agent-llm, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, embodied-agent, multi-agent, planning, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00334 diff --git a/papers/items/2026-2607-00345-registry-governed-agent-lifecycle-completing-eddops-with-evaluation-drivenregist.md b/papers/items/2026-2607-00345-registry-governed-agent-lifecycle-completing-eddops-with-evaluation-drivenregist.md new file mode 100644 index 0000000..fca4eff --- /dev/null +++ b/papers/items/2026-2607-00345-registry-governed-agent-lifecycle-completing-eddops-with-evaluation-drivenregist.md @@ -0,0 +1,60 @@ +# Paper: Registry-Governed Agent Lifecycle:Completing EDDOps with Evaluation-DrivenRegistration, Promotion, and Retirement on AWS AgentCore + +--- +type: paper +title: "Registry-Governed Agent Lifecycle:Completing EDDOps with Evaluation-DrivenRegistration, Promotion, and Retirement on AWS AgentCore" +authors: Richard Kang, Vincent Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00345 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, agent-safety, workflow-agent +- arXiv categories: cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00345 diff --git a/papers/items/2026-2607-00407-personalization-as-inverse-planning-learning-latent-design-intents-for-agentic-s.md b/papers/items/2026-2607-00407-personalization-as-inverse-planning-learning-latent-design-intents-for-agentic-s.md new file mode 100644 index 0000000..59c6f48 --- /dev/null +++ b/papers/items/2026-2607-00407-personalization-as-inverse-planning-learning-latent-design-intents-for-agentic-s.md @@ -0,0 +1,60 @@ +# Paper: Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising + +--- +type: paper +title: "Personalization as Inverse Planning: Learning Latent Design Intents for Agentic Slide Generation via Structural Denoising" +authors: Tianci Liu, Zihan Dong, Linjun Zhang, Haoyu Wang, jing Gao, Emre Kiciman, Ranveer Chandra, Wei-Ting Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00407 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - multi-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: multi-agent, planning, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00407 diff --git a/papers/items/2026-2607-00422-kidnaprag-a-black-box-attack-for-hijacking-reasoning-in-agentic-retrieval-augmen.md b/papers/items/2026-2607-00422-kidnaprag-a-black-box-attack-for-hijacking-reasoning-in-agentic-retrieval-augmen.md new file mode 100644 index 0000000..a4dedd0 --- /dev/null +++ b/papers/items/2026-2607-00422-kidnaprag-a-black-box-attack-for-hijacking-reasoning-in-agentic-retrieval-augmen.md @@ -0,0 +1,60 @@ +# Paper: KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems + +--- +type: paper +title: "KidnapRAG: A Black-Box Attack for Hijacking Reasoning in Agentic Retrieval-Augmented Generation Systems" +authors: Chanwoo Choi, Euntae Kim, Kyuho Lee, Youngsam Chun, Jinhee Jeong, Eunmi Kim, Myunggyo Oh, Junseo Jang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00422 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, rag, reasoning +- arXiv categories: cs.CR +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00422 diff --git a/papers/items/2026-2607-00436-phreeqc-mcq-200-a-diagnostic-benchmark-for-tool-augmented-scientific-simulator-a.md b/papers/items/2026-2607-00436-phreeqc-mcq-200-a-diagnostic-benchmark-for-tool-augmented-scientific-simulator-a.md new file mode 100644 index 0000000..fc1fa32 --- /dev/null +++ b/papers/items/2026-2607-00436-phreeqc-mcq-200-a-diagnostic-benchmark-for-tool-augmented-scientific-simulator-a.md @@ -0,0 +1,61 @@ +# Paper: PHREEQC-MCQ-200: A Diagnostic Benchmark for Tool-Augmented Scientific Simulator Agents + +--- +type: paper +title: "PHREEQC-MCQ-200: A Diagnostic Benchmark for Tool-Augmented Scientific Simulator Agents" +authors: Ke Zhang, Sahchit Chundur, Mohammad Javad Qomi, Maziar Raissi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00436 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, rag, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00436 diff --git a/papers/items/2026-2607-00440-minos-a-multi-agent-collaborative-framework-for-provenance-based-backward-tracki.md b/papers/items/2026-2607-00440-minos-a-multi-agent-collaborative-framework-for-provenance-based-backward-tracki.md new file mode 100644 index 0000000..f5fea34 --- /dev/null +++ b/papers/items/2026-2607-00440-minos-a-multi-agent-collaborative-framework-for-provenance-based-backward-tracki.md @@ -0,0 +1,63 @@ +# Paper: Minos: A Multi-Agent Collaborative Framework for Provenance-Based Backward Tracking + +--- +type: paper +title: "Minos: A Multi-Agent Collaborative Framework for Provenance-Based Backward Tracking" +authors: Jiahui Wang, Zhenyuan Li, Zhengkai Wang, Xiangmin Shen, Fan Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00440 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, rag, reasoning +- arXiv categories: cs.CR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00440 diff --git a/papers/items/2026-2607-00454-agri-sage-simulation-grounded-multi-agent-llm-for-context-aware-agricultural-adv.md b/papers/items/2026-2607-00454-agri-sage-simulation-grounded-multi-agent-llm-for-context-aware-agricultural-adv.md new file mode 100644 index 0000000..753a133 --- /dev/null +++ b/papers/items/2026-2607-00454-agri-sage-simulation-grounded-multi-agent-llm-for-context-aware-agricultural-adv.md @@ -0,0 +1,67 @@ +# Paper: Agri-SAGE: Simulation-Grounded Multi-Agent LLM for Context-Aware Agricultural Advisory Generation + +--- +type: paper +title: "Agri-SAGE: Simulation-Grounded Multi-Agent LLM for Context-Aware Agricultural Advisory Generation" +authors: Vedant Balasubramaniam, Geetha Charan, Manojkumar Patil, Rohit P Suresh, V Priyanka, Kodur Sai Vinay Sathvik, Y. Narahari +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00454 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - memory + - multi-agent + - planning + - rag + - reasoning + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, computer-use, memory, multi-agent, planning, rag, reasoning, world-model +- arXiv categories: cs.AI, cs.MA +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00454 diff --git a/papers/items/2026-2607-00502-a-task-state-representation-for-long-horizon-mobile-gui-agents.md b/papers/items/2026-2607-00502-a-task-state-representation-for-long-horizon-mobile-gui-agents.md new file mode 100644 index 0000000..6602547 --- /dev/null +++ b/papers/items/2026-2607-00502-a-task-state-representation-for-long-horizon-mobile-gui-agents.md @@ -0,0 +1,63 @@ +# Paper: A Task-State Representation for Long-Horizon Mobile GUI Agents + +--- +type: paper +title: A Task-State Representation for Long-Horizon Mobile GUI Agents +authors: Yujie Zheng, Zikang Liu, Xin Zhao, Ji-Rong Wen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00502 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, memory, planning, reasoning, tool-use +- arXiv categories: cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00502 diff --git a/papers/items/2026-2607-00555-rise-from-the-ashes-llm-based-static-analysis-for-deep-learning-framework-bugs.md b/papers/items/2026-2607-00555-rise-from-the-ashes-llm-based-static-analysis-for-deep-learning-framework-bugs.md new file mode 100644 index 0000000..8672415 --- /dev/null +++ b/papers/items/2026-2607-00555-rise-from-the-ashes-llm-based-static-analysis-for-deep-learning-framework-bugs.md @@ -0,0 +1,65 @@ +# Paper: Rise From The Ashes: LLM-based Static Analysis for Deep Learning Framework Bugs + +--- +type: paper +title: "Rise From The Ashes: LLM-based Static Analysis for Deep Learning Framework Bugs" +authors: Shaoyu Yang, Haifeng Lin, Chunrong Fang, Xiang Chen, Wei Cheng, Jiawei Liu, Yiyu Zhang, Hongyu Liu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00555 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - multi-agent + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, multi-agent, rag, tool-use, workflow-agent +- arXiv categories: cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00555 diff --git a/papers/items/2026-2607-00604-vehicle-routing-problem-meets-large-language-models-an-overview-and-perspectives.md b/papers/items/2026-2607-00604-vehicle-routing-problem-meets-large-language-models-an-overview-and-perspectives.md new file mode 100644 index 0000000..8dc5e68 --- /dev/null +++ b/papers/items/2026-2607-00604-vehicle-routing-problem-meets-large-language-models-an-overview-and-perspectives.md @@ -0,0 +1,63 @@ +# Paper: Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives + +--- +type: paper +title: "Vehicle Routing Problem Meets Large Language Models: An Overview and Perspectives" +authors: Xianchao Xiu, Chong Shen, Yanjiao Zhu, Wanquan Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00604 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - math.OC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, planning, reasoning, tool-use, workflow-agent +- arXiv categories: math.OC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00604 diff --git a/papers/items/2026-2607-00627-agi-maze-as-a-benchmark-framework-for-world-modeling-agents.md b/papers/items/2026-2607-00627-agi-maze-as-a-benchmark-framework-for-world-modeling-agents.md new file mode 100644 index 0000000..57588bb --- /dev/null +++ b/papers/items/2026-2607-00627-agi-maze-as-a-benchmark-framework-for-world-modeling-agents.md @@ -0,0 +1,61 @@ +# Paper: AGI Maze as a Benchmark Framework for World-Modeling Agents + +--- +type: paper +title: AGI Maze as a Benchmark Framework for World-Modeling Agents +authors: Alexey Potapov +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00627 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, memory, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00627 diff --git a/papers/items/2026-2607-00692-self-gc-self-governing-context-for-long-horizon-llm-agents.md b/papers/items/2026-2607-00692-self-gc-self-governing-context-for-long-horizon-llm-agents.md new file mode 100644 index 0000000..bac2426 --- /dev/null +++ b/papers/items/2026-2607-00692-self-gc-self-governing-context-for-long-horizon-llm-agents.md @@ -0,0 +1,60 @@ +# Paper: Self-GC: Self-Governing Context for Long-Horizon LLM Agents + +--- +type: paper +title: "Self-GC: Self-Governing Context for Long-Horizon LLM Agents" +authors: Xubin Hao, Hongjin Meng, Xin Yin, Jiawei Zhu, Chenpeng Cao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00692 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, planning-agent +- inferred topics: planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00692 diff --git a/papers/items/2026-2607-00911-from-registry-to-repository-how-ai-agent-skills-are-written-adapted-and-maintain.md b/papers/items/2026-2607-00911-from-registry-to-repository-how-ai-agent-skills-are-written-adapted-and-maintain.md new file mode 100644 index 0000000..f8c3d92 --- /dev/null +++ b/papers/items/2026-2607-00911-from-registry-to-repository-how-ai-agent-skills-are-written-adapted-and-maintain.md @@ -0,0 +1,60 @@ +# Paper: From Registry to Repository: How AI Agent Skills Are Written, Adapted, and Maintained + +--- +type: paper +title: "From Registry to Repository: How AI Agent Skills Are Written, Adapted, and Maintained" +authors: Haoyu Gao, Jai Lal Lulla, Hong Yi Lin, Sebastian Baltes, Christoph Treude, Mansooreh Zahedi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00911 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - coding-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: ai-agent, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, coding-agent +- inferred topics: coding-agent, tool-use, workflow-agent +- arXiv categories: cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00911 diff --git a/papers/items/2026-2607-00918-from-personas-to-plot-character-grounded-multi-agent-story-generation-for-long-f.md b/papers/items/2026-2607-00918-from-personas-to-plot-character-grounded-multi-agent-story-generation-for-long-f.md new file mode 100644 index 0000000..ec470cf --- /dev/null +++ b/papers/items/2026-2607-00918-from-personas-to-plot-character-grounded-multi-agent-story-generation-for-long-f.md @@ -0,0 +1,62 @@ +# Paper: From Personas to Plot: Character-Grounded Multi-Agent Story Generation for Long-Form Narratives + +--- +type: paper +title: "From Personas to Plot: Character-Grounded Multi-Agent Story Generation for Long-Form Narratives" +authors: Aayush Aluru, Chloe Ho, Muhammad Hammouri, Kerry Luo, Myra Malik, Ryan Lagasse, Arjun Bahuguna, Vasu Sharma +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00918 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, tool-use +- arXiv categories: cs.CL, cs.AI, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00918 diff --git a/papers/items/2026-2607-00939-leveraging-llm-based-agentic-systems-to-generate-quantum-applications-for-test-o.md b/papers/items/2026-2607-00939-leveraging-llm-based-agentic-systems-to-generate-quantum-applications-for-test-o.md new file mode 100644 index 0000000..b0d8985 --- /dev/null +++ b/papers/items/2026-2607-00939-leveraging-llm-based-agentic-systems-to-generate-quantum-applications-for-test-o.md @@ -0,0 +1,63 @@ +# Paper: Leveraging LLM-Based Agentic Systems to Generate Quantum Applications for Test Optimization + +--- +type: paper +title: Leveraging LLM-Based Agentic Systems to Generate Quantum Applications for Test Optimization +authors: Ming Tao, Yuechen Li, Tao Yue, Man Zhang, Aitor Arrieta Marcos +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00939 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - rag + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - quant-ph +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent, rag, workflow-agent +- arXiv categories: cs.SE, quant-ph +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00939 diff --git a/papers/items/2026-2607-00972-bayesian-uncertainty-propagation-for-agentic-rag-pipelines-a-proof-of-concept-st.md b/papers/items/2026-2607-00972-bayesian-uncertainty-propagation-for-agentic-rag-pipelines-a-proof-of-concept-st.md new file mode 100644 index 0000000..e303ae6 --- /dev/null +++ b/papers/items/2026-2607-00972-bayesian-uncertainty-propagation-for-agentic-rag-pipelines-a-proof-of-concept-st.md @@ -0,0 +1,62 @@ +# Paper: Bayesian Uncertainty Propagation for Agentic RAG Pipelines: A Proof-of-Concept Study on Multi-Hop Question Answering + +--- +type: paper +title: "Bayesian Uncertainty Propagation for Agentic RAG Pipelines: A Proof-of-Concept Study on Multi-Hop Question Answering" +authors: Louis Donaldson, Connor Walker, Koorosh Aslansefat, Yiannis Papadopoulos +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00972 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, workflow-agent +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00972 diff --git a/papers/items/2026-2607-00990-swe-doctor-guiding-software-engineering-agents-with-runtime-diagnosis-from-multi.md b/papers/items/2026-2607-00990-swe-doctor-guiding-software-engineering-agents-with-runtime-diagnosis-from-multi.md new file mode 100644 index 0000000..68825dc --- /dev/null +++ b/papers/items/2026-2607-00990-swe-doctor-guiding-software-engineering-agents-with-runtime-diagnosis-from-multi.md @@ -0,0 +1,62 @@ +# Paper: SWE-Doctor: Guiding Software Engineering Agents with Runtime Diagnosis from Multi-Faceted Bug Reproduction Tests + +--- +type: paper +title: "SWE-Doctor: Guiding Software Engineering Agents with Runtime Diagnosis from Multi-Faceted Bug Reproduction Tests" +authors: Yaoqi Guo, Yang Liu, Jie M. Zhang, Yun Ma, Yiling Lou, Zhenpeng Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.00990 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, rag +- arXiv categories: cs.SE, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.00990 diff --git a/papers/items/2026-2607-01047-conversable-complexity-agentic-llm-collectives-as-interpretable-substrates.md b/papers/items/2026-2607-01047-conversable-complexity-agentic-llm-collectives-as-interpretable-substrates.md new file mode 100644 index 0000000..dcd2125 --- /dev/null +++ b/papers/items/2026-2607-01047-conversable-complexity-agentic-llm-collectives-as-interpretable-substrates.md @@ -0,0 +1,59 @@ +# Paper: Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates + +--- +type: paper +title: "Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates" +authors: Elias Najarro, Ane Espeseth, Eleni Nisioti, Sebastian Risi, Stefano Nichele +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01047 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: memory, tool-use +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01047 diff --git a/papers/items/2026-2607-01061-agentic-generation-of-verifiable-rules-for-deterministic-self-expanding-reaction.md b/papers/items/2026-2607-01061-agentic-generation-of-verifiable-rules-for-deterministic-self-expanding-reaction.md new file mode 100644 index 0000000..7b8af78 --- /dev/null +++ b/papers/items/2026-2607-01061-agentic-generation-of-verifiable-rules-for-deterministic-self-expanding-reaction.md @@ -0,0 +1,62 @@ +# Paper: Agentic generation of verifiable rules for deterministic, self-expanding reaction classification + +--- +type: paper +title: Agentic generation of verifiable rules for deterministic, self-expanding reaction classification +authors: Daniel Armstrong, Maarten Dobbelaere, Valentas Olikauskas, Helena Avila, Octavian Susanu, Jérôme Waser, Philippe Schwaller +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01061 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - coding-agent + - multi-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: coding-agent, multi-agent, planning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01061 diff --git a/papers/items/2026-2607-01071-memsyco-bench-benchmarking-sycophancy-in-agent-memory.md b/papers/items/2026-2607-01071-memsyco-bench-benchmarking-sycophancy-in-agent-memory.md new file mode 100644 index 0000000..f7c87c5 --- /dev/null +++ b/papers/items/2026-2607-01071-memsyco-bench-benchmarking-sycophancy-in-agent-memory.md @@ -0,0 +1,62 @@ +# Paper: MemSyco-Bench: Benchmarking Sycophancy in Agent Memory + +--- +type: paper +title: "MemSyco-Bench: Benchmarking Sycophancy in Agent Memory" +authors: Zhishang Xiang, Zerui Chen, Yunbo Tang, Zhimin Wei, Ruqin Ning, Yujie Lin, Qinggang Zhang, Jinsong Su +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01071 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, memory, reasoning +- arXiv categories: cs.IR, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01071 diff --git a/papers/items/2026-2607-01084-can-agents-generalize-to-the-open-world-unveiling-the-fragility-of-static-traini.md b/papers/items/2026-2607-01084-can-agents-generalize-to-the-open-world-unveiling-the-fragility-of-static-traini.md new file mode 100644 index 0000000..0cf7b1f --- /dev/null +++ b/papers/items/2026-2607-01084-can-agents-generalize-to-the-open-world-unveiling-the-fragility-of-static-traini.md @@ -0,0 +1,61 @@ +# Paper: Can Agents Generalize to the Open World? Unveiling the Fragility of Static Training in Tool Use + +--- +type: paper +title: Can Agents Generalize to the Open World? Unveiling the Fragility of Static Training in Tool Use +authors: Song-Lin Lv, Weiming Wu, Rui Zhu, Zi-Jian Cheng, Lan-Zhe Guo +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01084 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, tool-use +- inferred topics: agent-evaluation, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01084 diff --git a/papers/items/2026-2607-01120-next-generation-agentic-reinforcement-learning-systems-enable-self-evolving-agen.md b/papers/items/2026-2607-01120-next-generation-agentic-reinforcement-learning-systems-enable-self-evolving-agen.md new file mode 100644 index 0000000..1005c0e --- /dev/null +++ b/papers/items/2026-2607-01120-next-generation-agentic-reinforcement-learning-systems-enable-self-evolving-agen.md @@ -0,0 +1,61 @@ +# Paper: Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents + +--- +type: paper +title: Next-Generation Agentic Reinforcement Learning Systems Enable Self-Evolving Agents +authors: Ran Yan, Wei Fu, Jiale Li, Shusheng Xu, Zhiyu Mei, Jiaxuan Gao, Jiarui Zhang, Wentai Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01120 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - coding-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: coding-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.DC +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01120 diff --git a/papers/items/2026-2607-01211-are-performance-optimization-benchmarks-reliably-measuring-coding-agents.md b/papers/items/2026-2607-01211-are-performance-optimization-benchmarks-reliably-measuring-coding-agents.md new file mode 100644 index 0000000..b4d67bc --- /dev/null +++ b/papers/items/2026-2607-01211-are-performance-optimization-benchmarks-reliably-measuring-coding-agents.md @@ -0,0 +1,61 @@ +# Paper: Are Performance-Optimization Benchmarks Reliably Measuring Coding Agents? + +--- +type: paper +title: Are Performance-Optimization Benchmarks Reliably Measuring Coding Agents? +authors: Zhi Chen, Zhensu Sun, Yuling Shi, David Lo, Lingxiao Jiang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01211 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, rag +- arXiv categories: cs.SE, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01211 diff --git a/papers/items/2026-2607-01213-reporescue-an-empirical-study-of-llm-agents-on-whole-repository-compatibility-re.md b/papers/items/2026-2607-01213-reporescue-an-empirical-study-of-llm-agents-on-whole-repository-compatibility-re.md new file mode 100644 index 0000000..a782bdc --- /dev/null +++ b/papers/items/2026-2607-01213-reporescue-an-empirical-study-of-llm-agents-on-whole-repository-compatibility-re.md @@ -0,0 +1,61 @@ +# Paper: RepoRescue: An Empirical Study of LLM Agents on Whole-Repository Compatibility Rescue + +--- +type: paper +title: "RepoRescue: An Empirical Study of LLM Agents on Whole-Repository Compatibility Rescue" +authors: Zhihao Lin, Mingyi Zhou, Zhensu Sun, Yizhuo Yang, Renyu Yang, David Lo, Li Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01213 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, coding-agent, reasoning, tool-use +- arXiv categories: cs.SE +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01213 diff --git a/papers/items/2026-2607-01366-auto-fl-research-agentic-search-for-federated-learning-algorithms.md b/papers/items/2026-2607-01366-auto-fl-research-agentic-search-for-federated-learning-algorithms.md new file mode 100644 index 0000000..b726e3c --- /dev/null +++ b/papers/items/2026-2607-01366-auto-fl-research-agentic-search-for-federated-learning-algorithms.md @@ -0,0 +1,60 @@ +# Paper: Auto-FL-Research: Agentic Search for Federated Learning Algorithms + +--- +type: paper +title: "Auto-FL-Research: Agentic Search for Federated Learning Algorithms" +authors: Holger R. Roth, Ziyue Xu, Chester Chen, Daguang Xu, Peter Cnudde, Andrew Feng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01366 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agentic-ai, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, coding-agent +- inferred topics: agent-evaluation, coding-agent, workflow-agent +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01366 diff --git a/papers/items/2026-2607-01421-risk-architecture-for-ai-native-engineering-teams-an-organizational-framework-fo.md b/papers/items/2026-2607-01421-risk-architecture-for-ai-native-engineering-teams-an-organizational-framework-fo.md new file mode 100644 index 0000000..f365104 --- /dev/null +++ b/papers/items/2026-2607-01421-risk-architecture-for-ai-native-engineering-teams-an-organizational-framework-fo.md @@ -0,0 +1,64 @@ +# Paper: Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance + +--- +type: paper +title: "Risk Architecture for AI-Native Engineering Teams: An Organizational Framework for Agentic System Governance" +authors: Laxmipriya Ganesh Iyer +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01421 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, rag, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01421 diff --git a/papers/items/2026-2607-01510-janus-a-playground-for-user-involved-agentic-permission-management.md b/papers/items/2026-2607-01510-janus-a-playground-for-user-involved-agentic-permission-management.md new file mode 100644 index 0000000..ca5793c --- /dev/null +++ b/papers/items/2026-2607-01510-janus-a-playground-for-user-involved-agentic-permission-management.md @@ -0,0 +1,61 @@ +# Paper: Janus: a Playground for User-Involved Agentic Permission Management + +--- +type: paper +title: "Janus: a Playground for User-Involved Agentic Permission Management" +authors: Natalie Grace Brigham, Eugene Bagdasarian, Tadayoshi Kohno, Franziska Roesner +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01510 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.AI, cs.CR +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01510 diff --git a/papers/items/2026-2607-01523-multi-head-recurrent-memory-agents.md b/papers/items/2026-2607-01523-multi-head-recurrent-memory-agents.md new file mode 100644 index 0000000..8d8e6d1 --- /dev/null +++ b/papers/items/2026-2607-01523-multi-head-recurrent-memory-agents.md @@ -0,0 +1,62 @@ +# Paper: Multi-Head Recurrent Memory Agents + +--- +type: paper +title: Multi-Head Recurrent Memory Agents +authors: Jiatong Li, Samuel Yeh, Sharon Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01523 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, agent-safety, memory +- arXiv categories: cs.LG, cs.AI, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01523 diff --git a/papers/items/2026-2607-01531-opine-world-programmatic-world-modeling-with-ontology-error-prioritized-interact.md b/papers/items/2026-2607-01531-opine-world-programmatic-world-modeling-with-ontology-error-prioritized-interact.md new file mode 100644 index 0000000..46304b4 --- /dev/null +++ b/papers/items/2026-2607-01531-opine-world-programmatic-world-modeling-with-ontology-error-prioritized-interact.md @@ -0,0 +1,63 @@ +# Paper: OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration + +--- +type: paper +title: "OPINE-World: Programmatic World Modeling with Ontology-error-Prioritized Interactive Exploration" +authors: David Courtis, Wenhao Li, Scott Sanner +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01531 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: llm-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, planning-agent +- inferred topics: agent-evaluation, computer-use, planning, tool-use, world-model +- arXiv categories: cs.AI, cs.LG +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01531 diff --git a/papers/items/2026-2607-01600-boundary-sync-measuring-communication-induced-representational-coupling-in-multi.md b/papers/items/2026-2607-01600-boundary-sync-measuring-communication-induced-representational-coupling-in-multi.md new file mode 100644 index 0000000..8646e48 --- /dev/null +++ b/papers/items/2026-2607-01600-boundary-sync-measuring-communication-induced-representational-coupling-in-multi.md @@ -0,0 +1,60 @@ +# Paper: BOUNDARY_SYNC: Measuring Communication-Induced Representational Coupling in Multi-Agent LLM Systems + +--- +type: paper +title: "BOUNDARY_SYNC: Measuring Communication-Induced Representational Coupling in Multi-Agent LLM Systems" +authors: Zewen Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01600 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: multi-agent, tool-use +- arXiv categories: cs.LG, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01600 diff --git a/papers/items/2026-2607-01640-agentflow-building-agent-dependency-graphs-for-static-analysis-of-agent-programs.md b/papers/items/2026-2607-01640-agentflow-building-agent-dependency-graphs-for-static-analysis-of-agent-programs.md new file mode 100644 index 0000000..e5be2b6 --- /dev/null +++ b/papers/items/2026-2607-01640-agentflow-building-agent-dependency-graphs-for-static-analysis-of-agent-programs.md @@ -0,0 +1,63 @@ +# Paper: AgentFlow: Building Agent Dependency Graphs for Static Analysis of Agent Programs + +--- +type: paper +title: "AgentFlow: Building Agent Dependency Graphs for Static Analysis of Agent Programs" +authors: Shenao Wang, Xinyi Hou, Yanjie Zhao, Xiao Cheng, Haoyu Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01640 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, memory, multi-agent, tool-use +- arXiv categories: cs.SE, cs.CR +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01640 diff --git a/papers/items/2026-2607-01641-when-agents-do-not-stop-uncovering-infinite-agentic-loops-in-llm-agents.md b/papers/items/2026-2607-01641-when-agents-do-not-stop-uncovering-infinite-agentic-loops-in-llm-agents.md new file mode 100644 index 0000000..db02a8a --- /dev/null +++ b/papers/items/2026-2607-01641-when-agents-do-not-stop-uncovering-infinite-agentic-loops-in-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: When Agents Do Not Stop: Uncovering Infinite Agentic Loops in LLM Agents + +--- +type: paper +title: "When Agents Do Not Stop: Uncovering Infinite Agentic Loops in LLM Agents" +authors: Xinyi Hou, Shenao Wang, Yanjie Zhao, Haoyu Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01641 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: llm-agent, planning-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, planning-agent, tool-use +- inferred topics: agent-evaluation, multi-agent, planning, tool-use, workflow-agent +- arXiv categories: cs.SE +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01641 diff --git a/papers/items/2026-2607-01661-diverse-evidence-better-forecasts-multi-agent-deliberation-under-information-asy.md b/papers/items/2026-2607-01661-diverse-evidence-better-forecasts-multi-agent-deliberation-under-information-asy.md new file mode 100644 index 0000000..500ae23 --- /dev/null +++ b/papers/items/2026-2607-01661-diverse-evidence-better-forecasts-multi-agent-deliberation-under-information-asy.md @@ -0,0 +1,60 @@ +# Paper: Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry + +--- +type: paper +title: "Diverse Evidence, Better Forecasts: Multi-Agent Deliberation Under Information Asymmetry" +authors: Yuante Li, Yicheng Tao, Kate Zhang, Taozhi Wang, Gefei Gu, Yaxin Zhou +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01661 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, reasoning +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01661 diff --git a/papers/items/2026-2607-01668-verichat-an-agentic-conversational-ai-assistant-for-hardware-security-verificati.md b/papers/items/2026-2607-01668-verichat-an-agentic-conversational-ai-assistant-for-hardware-security-verificati.md new file mode 100644 index 0000000..a098d6b --- /dev/null +++ b/papers/items/2026-2607-01668-verichat-an-agentic-conversational-ai-assistant-for-hardware-security-verificati.md @@ -0,0 +1,65 @@ +# Paper: VeriChat: An Agentic Conversational AI Assistant for Hardware Security Verification + +--- +type: paper +title: "VeriChat: An Agentic Conversational AI Assistant for Hardware Security Verification" +authors: Dipayan Saha, Khan Thamid Hasan, Shams Tarek, Sujan Kumar Saha, Mark Tehranipoor, Farimah Farahmandi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01668 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - multi-agent + - rag + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, agent-safety, computer-use, multi-agent, rag, tool-use, workflow-agent, world-model +- arXiv categories: cs.CR +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01668 diff --git a/papers/items/2026-2607-01709-comfyclaw-self-evolving-skill-harnesses-for-image-generation-workflows.md b/papers/items/2026-2607-01709-comfyclaw-self-evolving-skill-harnesses-for-image-generation-workflows.md new file mode 100644 index 0000000..d4c11ac --- /dev/null +++ b/papers/items/2026-2607-01709-comfyclaw-self-evolving-skill-harnesses-for-image-generation-workflows.md @@ -0,0 +1,64 @@ +# Paper: COMFYCLAW: Self-Evolving Skill Harnesses for Image Generation Workflows + +--- +type: paper +title: "COMFYCLAW: Self-Evolving Skill Harnesses for Image Generation Workflows" +authors: Zongxia Li, Dawei Liu, Fuxiao Liu, Yuhang Zhou, Xiyang Wu, Jingxi Chen, Jing Xie, Xiaomin Wu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01709 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01709 diff --git a/papers/items/2026-2607-01766-simworlds-a-multi-agent-system-for-dynamic-3d-scene-creation.md b/papers/items/2026-2607-01766-simworlds-a-multi-agent-system-for-dynamic-3d-scene-creation.md new file mode 100644 index 0000000..56fe884 --- /dev/null +++ b/papers/items/2026-2607-01766-simworlds-a-multi-agent-system-for-dynamic-3d-scene-creation.md @@ -0,0 +1,64 @@ +# Paper: SimWorlds: A Multi-Agent System for Dynamic 3D Scene Creation + +--- +type: paper +title: "SimWorlds: A Multi-Agent System for Dynamic 3D Scene Creation" +authors: Chunjiang Liu, Xiaoyuan Wang, Haoyu Chen, Yizhou Zhao, Ming-Hsuan Yang, László A. Jeni +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01766 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - multi-agent + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, embodied-agent, multi-agent, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01766 diff --git a/papers/items/2026-2607-01767-repair-the-amplifier-not-the-symptom-stable-world-model-correction-for-agent-rol.md b/papers/items/2026-2607-01767-repair-the-amplifier-not-the-symptom-stable-world-model-correction-for-agent-rol.md new file mode 100644 index 0000000..09e2cf7 --- /dev/null +++ b/papers/items/2026-2607-01767-repair-the-amplifier-not-the-symptom-stable-world-model-correction-for-agent-rol.md @@ -0,0 +1,62 @@ +# Paper: Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts + +--- +type: paper +title: "Repair the Amplifier, Not the Symptom: Stable World-Model Correction for Agent Rollouts" +authors: Xinyuan Song, Zekun Cai +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01767 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, planning, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01767 diff --git a/papers/items/2026-2607-01788-krca-an-efficient-root-cause-analysis-system-in-hyper-scale-microservice-systems.md b/papers/items/2026-2607-01788-krca-an-efficient-root-cause-analysis-system-in-hyper-scale-microservice-systems.md new file mode 100644 index 0000000..00f8b2c --- /dev/null +++ b/papers/items/2026-2607-01788-krca-an-efficient-root-cause-analysis-system-in-hyper-scale-microservice-systems.md @@ -0,0 +1,63 @@ +# Paper: KRCA: An Efficient Root Cause Analysis System in Hyper-Scale Microservice Systems via Agentic AI + +--- +type: paper +title: "KRCA: An Efficient Root Cause Analysis System in Hyper-Scale Microservice Systems via Agentic AI" +authors: Jiamin Jiang, Jingfei Feng, Yu Luo, Qingliang Zhang, Yongqian Su, Wenwei Gu, Shenglin Zhang, Tianyu Cui, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01788 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, memory, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01788 diff --git a/papers/items/2026-2607-01793-safety-testing-llm-agents-at-scale-from-risk-discovery-to-evidence-grounded-veri.md b/papers/items/2026-2607-01793-safety-testing-llm-agents-at-scale-from-risk-discovery-to-evidence-grounded-veri.md new file mode 100644 index 0000000..45f28de --- /dev/null +++ b/papers/items/2026-2607-01793-safety-testing-llm-agents-at-scale-from-risk-discovery-to-evidence-grounded-veri.md @@ -0,0 +1,63 @@ +# Paper: Safety Testing LLM Agents at Scale: From Risk Discovery to Evidence-Grounded Verification + +--- +type: paper +title: "Safety Testing LLM Agents at Scale: From Risk Discovery to Evidence-Grounded Verification" +authors: Yunhao Feng, Ruixiao Lin, Ming Wen, Qinqin He, Yanming Guo, Yifan Ding, Yutao Wu, Jialuo Chen, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01793 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01793 diff --git a/papers/items/2026-2607-01812-to-master-an-llm-agent-framework-for-automated-topology-optimization.md b/papers/items/2026-2607-01812-to-master-an-llm-agent-framework-for-automated-topology-optimization.md new file mode 100644 index 0000000..bcfc8c9 --- /dev/null +++ b/papers/items/2026-2607-01812-to-master-an-llm-agent-framework-for-automated-topology-optimization.md @@ -0,0 +1,62 @@ +# Paper: TO-Master: an LLM-agent framework for automated topology optimization + +--- +type: paper +title: "TO-Master: an LLM-agent framework for automated topology optimization" +authors: Haoju Lin, Wenchang Zhang, Weipeng Xu, Xiang Li, Tian Xu, Tianju Xue +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01812 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, computer-use, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CE +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01812 diff --git a/papers/items/2026-2607-01874-skillcoach-self-evolving-rubrics-for-evaluating-and-enhancing-agentic-skill-use.md b/papers/items/2026-2607-01874-skillcoach-self-evolving-rubrics-for-evaluating-and-enhancing-agentic-skill-use.md new file mode 100644 index 0000000..c73adba --- /dev/null +++ b/papers/items/2026-2607-01874-skillcoach-self-evolving-rubrics-for-evaluating-and-enhancing-agentic-skill-use.md @@ -0,0 +1,64 @@ +# Paper: SkillCoach: Self-Evolving Rubrics for Evaluating and Enhancing Agentic Skill-Use + +--- +type: paper +title: "SkillCoach: Self-Evolving Rubrics for Evaluating and Enhancing Agentic Skill-Use" +authors: Jiayin Zhu, Kelong Mao, Yudong Guo, Dengbo He, Sulong Xu, Simiu Gu, Yutao Yue +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01874 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CL +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01874 diff --git a/papers/items/2026-2607-01916-contextsniper-anttrail-s-token-efficient-code-memory-for-repository-level-progra.md b/papers/items/2026-2607-01916-contextsniper-anttrail-s-token-efficient-code-memory-for-repository-level-progra.md new file mode 100644 index 0000000..8fa37d0 --- /dev/null +++ b/papers/items/2026-2607-01916-contextsniper-anttrail-s-token-efficient-code-memory-for-repository-level-progra.md @@ -0,0 +1,62 @@ +# Paper: ContextSniper: AntTrail's Token-Efficient Code Memory for Repository-Level Program Repair + +--- +type: paper +title: "ContextSniper: AntTrail's Token-Efficient Code Memory for Repository-Level Program Repair" +authors: Chiwang Luk, Matin Mohammad Najafi, Zhifeng Jia, Wei Yang, Xiuchang Li, Jinwei Zhu, Yang Ren, Lei Chen, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01916 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory, coding-agent, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, coding-agent, rag-agent +- inferred topics: agent-evaluation, coding-agent, memory, rag, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01916 diff --git a/papers/items/2026-2607-01929-beyond-textual-repository-exploration-dual-modal-structural-reasoning-for-agenti.md b/papers/items/2026-2607-01929-beyond-textual-repository-exploration-dual-modal-structural-reasoning-for-agenti.md new file mode 100644 index 0000000..573ad61 --- /dev/null +++ b/papers/items/2026-2607-01929-beyond-textual-repository-exploration-dual-modal-structural-reasoning-for-agenti.md @@ -0,0 +1,64 @@ +# Paper: Beyond Textual Repository Exploration: Dual-Modal Structural Reasoning for Agentic Issue Resolution + +--- +type: paper +title: "Beyond Textual Repository Exploration: Dual-Modal Structural Reasoning for Agentic Issue Resolution" +authors: Jiayi Zhang, Kai Huang, Yang Liu, Chunyang Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01929 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - embodied-agent + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: coding-agent, function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, function-calling +- inferred topics: agent-evaluation, coding-agent, embodied-agent, planning, rag, reasoning, tool-use +- arXiv categories: cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01929 diff --git a/papers/items/2026-2607-01935-a-tma-decoupling-state-aware-memory-failures-in-long-term-agent-memory.md b/papers/items/2026-2607-01935-a-tma-decoupling-state-aware-memory-failures-in-long-term-agent-memory.md new file mode 100644 index 0000000..7fdd37a --- /dev/null +++ b/papers/items/2026-2607-01935-a-tma-decoupling-state-aware-memory-failures-in-long-term-agent-memory.md @@ -0,0 +1,60 @@ +# Paper: A-TMA: Decoupling State-Aware Memory Failures in Long-Term Agent Memory + +--- +type: paper +title: "A-TMA: Decoupling State-Aware Memory Failures in Long-Term Agent Memory" +authors: Zitong Shi, Yixuan Tang, Anthony Kum Hoe Tung +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.01935 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-memory, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, llm-agent +- inferred topics: agent-evaluation, memory, rag +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.01935 diff --git a/papers/items/2026-2607-02032-pace-a-proxy-for-agentic-capability-evaluation.md b/papers/items/2026-2607-02032-pace-a-proxy-for-agentic-capability-evaluation.md new file mode 100644 index 0000000..2560228 --- /dev/null +++ b/papers/items/2026-2607-02032-pace-a-proxy-for-agentic-capability-evaluation.md @@ -0,0 +1,61 @@ +# Paper: PACE: A Proxy for Agentic Capability Evaluation + +--- +type: paper +title: "PACE: A Proxy for Agentic Capability Evaluation" +authors: Yueqi Song, Lintang Sutawika, Jiarui Liu, Lindia Tjuatja, Jiayi Geng, Yunze Xiao, Daniel Lee, Aditya Bharat Soni, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02032 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: agent-evaluation, coding-agent, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, coding-agent, llm-agent +- inferred topics: agent-evaluation, coding-agent, reasoning +- arXiv categories: cs.AI, cs.CL +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02032 diff --git a/papers/items/2026-2607-02186-ua-chatdev-uncertainty-aware-multi-agent-collaboration-for-reliable-software-dev.md b/papers/items/2026-2607-02186-ua-chatdev-uncertainty-aware-multi-agent-collaboration-for-reliable-software-dev.md new file mode 100644 index 0000000..96a543d --- /dev/null +++ b/papers/items/2026-2607-02186-ua-chatdev-uncertainty-aware-multi-agent-collaboration-for-reliable-software-dev.md @@ -0,0 +1,62 @@ +# Paper: UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development + +--- +type: paper +title: "UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development" +authors: Temitayo Olamilekan Ogunsusi, Lijun Qian, Xishuang Dong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02186 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent, rag, tool-use +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02186 diff --git a/papers/items/2026-2607-02210-criticality-based-guard-rail-validation-for-ai-agent-decisions-in-autonomous-tel.md b/papers/items/2026-2607-02210-criticality-based-guard-rail-validation-for-ai-agent-decisions-in-autonomous-tel.md new file mode 100644 index 0000000..45af6ce --- /dev/null +++ b/papers/items/2026-2607-02210-criticality-based-guard-rail-validation-for-ai-agent-decisions-in-autonomous-tel.md @@ -0,0 +1,63 @@ +# Paper: Criticality-Based Guard Rail Validation for AI Agent Decisions in Autonomous Telecom Networks + +--- +type: paper +title: Criticality-Based Guard Rail Validation for AI Agent Decisions in Autonomous Telecom Networks +authors: Ravi Kant Sharma +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02210 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.NI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, multi-agent, rag, tool-use +- arXiv categories: cs.AI, cs.NI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02210 diff --git a/papers/items/2026-2607-02245-copewell-a-multi-agent-swarm-architecture-for-equitable-mental-wellness-support.md b/papers/items/2026-2607-02245-copewell-a-multi-agent-swarm-architecture-for-equitable-mental-wellness-support.md new file mode 100644 index 0000000..e1e2817 --- /dev/null +++ b/papers/items/2026-2607-02245-copewell-a-multi-agent-swarm-architecture-for-equitable-mental-wellness-support.md @@ -0,0 +1,62 @@ +# Paper: Copewell: A Multi-Agent Swarm Architecture for Equitable Mental Wellness Support + +--- +type: paper +title: "Copewell: A Multi-Agent Swarm Architecture for Equitable Mental Wellness Support" +authors: Seren Yenikent, Jack Vinijtrongjit, Katherine Ng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02245 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CY + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, multi-agent +- arXiv categories: cs.AI, cs.CY, cs.HC +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02245 diff --git a/papers/items/2026-2607-02255-agenticsts-a-bounded-memory-testbed-for-long-horizon-llm-agents.md b/papers/items/2026-2607-02255-agenticsts-a-bounded-memory-testbed-for-long-horizon-llm-agents.md new file mode 100644 index 0000000..a4e5ce9 --- /dev/null +++ b/papers/items/2026-2607-02255-agenticsts-a-bounded-memory-testbed-for-long-horizon-llm-agents.md @@ -0,0 +1,64 @@ +# Paper: AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents + +--- +type: paper +title: "AgenticSTS: A Bounded-Memory Testbed for Long-Horizon LLM Agents" +authors: Xiangchen Cheng, Yunwei Jiang, Jianwen Sun, Zizhen Li, Chuanhao Li, Xiangcheng Cao, Yihao Liu, Fanrui Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02255 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 23 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 23 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02255 diff --git a/papers/items/2026-2607-02294-coding-agents-are-guessing-measuring-action-boundary-violations-in-underspecifie.md b/papers/items/2026-2607-02294-coding-agents-are-guessing-measuring-action-boundary-violations-in-underspecifie.md new file mode 100644 index 0000000..2138d3a --- /dev/null +++ b/papers/items/2026-2607-02294-coding-agents-are-guessing-measuring-action-boundary-violations-in-underspecifie.md @@ -0,0 +1,61 @@ +# Paper: Coding Agents Are Guessing: Measuring Action-Boundary Violations in Underspecified DevOps Instructions + +--- +type: paper +title: "Coding Agents Are Guessing: Measuring Action-Boundary Violations in Underspecified DevOps Instructions" +authors: Zimo Ji, Zekai Zhang, Congying Xu, Zongjie Li, Yudong Gao, Shuai Wang, Shing-Chi Cheung +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02294 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02294 diff --git a/papers/items/2026-2607-02381-hulat2-at-mer-trans-2026-governed-multi-agent-simplification-for-spanish-easy-to.md b/papers/items/2026-2607-02381-hulat2-at-mer-trans-2026-governed-multi-agent-simplification-for-spanish-easy-to.md new file mode 100644 index 0000000..4037037 --- /dev/null +++ b/papers/items/2026-2607-02381-hulat2-at-mer-trans-2026-governed-multi-agent-simplification-for-spanish-easy-to.md @@ -0,0 +1,62 @@ +# Paper: HULAT2 at MER-TRANS 2026: Governed Multi-Agent Simplification for Spanish Easy-to-Read Generation + +--- +type: paper +title: "HULAT2 at MER-TRANS 2026: Governed Multi-Agent Simplification for Spanish Easy-to-Read Generation" +authors: Lourdes Moreno, Paloma Martínez, Marco Antonio Sanchez-Escudero, Miguel Domínguez-Gómez +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02381 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - multi-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, computer-use, multi-agent, tool-use, workflow-agent +- arXiv categories: cs.CL +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02381 diff --git a/papers/items/2026-2607-02448-agentscad-automated-design-for-manufacturing-of-fdm-parts-via-multi-agent-llm-re.md b/papers/items/2026-2607-02448-agentscad-automated-design-for-manufacturing-of-fdm-parts-via-multi-agent-llm-re.md new file mode 100644 index 0000000..4684a53 --- /dev/null +++ b/papers/items/2026-2607-02448-agentscad-automated-design-for-manufacturing-of-fdm-parts-via-multi-agent-llm-re.md @@ -0,0 +1,61 @@ +# Paper: AgentsCAD: Automated Design for Manufacturing of FDM Parts via Multi-Agent LLM Reasoning and Geometric Feature Recognition + +--- +type: paper +title: "AgentsCAD: Automated Design for Manufacturing of FDM Parts via Multi-Agent LLM Reasoning and Geometric Feature Recognition" +authors: Emmanuel George, Christopher Keefe, Peter Pak, Amir Barati Farimani +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02448 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, reasoning, workflow-agent +- arXiv categories: cs.MA +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02448 diff --git a/papers/items/2026-2607-02453-adoption-and-ecosystem-health-a-longitudinal-analysis-of-open-source-multi-agent.md b/papers/items/2026-2607-02453-adoption-and-ecosystem-health-a-longitudinal-analysis-of-open-source-multi-agent.md new file mode 100644 index 0000000..c9c5664 --- /dev/null +++ b/papers/items/2026-2607-02453-adoption-and-ecosystem-health-a-longitudinal-analysis-of-open-source-multi-agent.md @@ -0,0 +1,59 @@ +# Paper: Adoption and Ecosystem Health: A Longitudinal Analysis of Open-Source Multi-Agent Frameworks + +--- +type: paper +title: "Adoption and Ecosystem Health: A Longitudinal Analysis of Open-Source Multi-Agent Frameworks" +authors: Xi Zhang, Papi Menon, Vivian Chu, Koray Cosguner +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02453 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, multi-agent +- arXiv categories: cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02453 diff --git a/papers/items/2026-2607-02507-what-llm-agents-say-when-no-one-is-watching-social-structure-and-latent-objectiv.md b/papers/items/2026-2607-02507-what-llm-agents-say-when-no-one-is-watching-social-structure-and-latent-objectiv.md new file mode 100644 index 0000000..b5034d5 --- /dev/null +++ b/papers/items/2026-2607-02507-what-llm-agents-say-when-no-one-is-watching-social-structure-and-latent-objectiv.md @@ -0,0 +1,63 @@ +# Paper: What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates + +--- +type: paper +title: "What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates" +authors: Arman Ghaffarizadeh, Danyal Mohaddes, Aliakbar Izadkhah, Shahriar Noroozizadeh +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02507 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation, llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent +- arXiv categories: cs.AI, cs.CL, cs.LG, cs.MA +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02507 diff --git a/papers/items/2026-2607-02577-benchmarking-the-benchmarks-a-validity-audit-of-tool-calling-evaluation.md b/papers/items/2026-2607-02577-benchmarking-the-benchmarks-a-validity-audit-of-tool-calling-evaluation.md new file mode 100644 index 0000000..22207ea --- /dev/null +++ b/papers/items/2026-2607-02577-benchmarking-the-benchmarks-a-validity-audit-of-tool-calling-evaluation.md @@ -0,0 +1,60 @@ +# Paper: Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation + +--- +type: paper +title: "Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation" +authors: Vishvesh Bhat, Jay Vaghasiya, Muhammad Ahmed Mohsin, Asad Aali +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02577 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.SE +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02577 diff --git a/papers/items/2026-2607-02579-when-not-to-write-memory-governing-false-promotion-from-correlated-agent-traces.md b/papers/items/2026-2607-02579-when-not-to-write-memory-governing-false-promotion-from-correlated-agent-traces.md new file mode 100644 index 0000000..15dc234 --- /dev/null +++ b/papers/items/2026-2607-02579-when-not-to-write-memory-governing-false-promotion-from-correlated-agent-traces.md @@ -0,0 +1,63 @@ +# Paper: When Not to Write Memory: Governing False Promotion from Correlated Agent Traces + +--- +type: paper +title: "When Not to Write Memory: Governing False Promotion from Correlated Agent Traces" +authors: Yijiashun Qi, Xiang Xu, Yuxuan Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02579 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-06-30 +updated_at: 2026-06-30 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-memory, language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, language-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, memory, rag, tool-use +- arXiv categories: cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02579 diff --git a/papers/items/2026-2607-02599-agentltl-a-trace-verification-framework-for-measuring-enforcing-and-training-pro.md b/papers/items/2026-2607-02599-agentltl-a-trace-verification-framework-for-measuring-enforcing-and-training-pro.md new file mode 100644 index 0000000..9fe1914 --- /dev/null +++ b/papers/items/2026-2607-02599-agentltl-a-trace-verification-framework-for-measuring-enforcing-and-training-pro.md @@ -0,0 +1,63 @@ +# Paper: AgentLTL: A Trace-Verification Framework for Measuring, Enforcing, and Training Procedural Compliance in Tool-Using LLM Agents + +--- +type: paper +title: "AgentLTL: A Trace-Verification Framework for Measuring, Enforcing, and Training Procedural Compliance in Tool-Using LLM Agents" +authors: Laïla Elkoussy, Julien Perez +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02599 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.LO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, tool-use +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: cs.SE, cs.AI, cs.LO +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02599 diff --git a/papers/items/2026-2607-02606-chainswe-benchmarking-coding-agents-on-multi-bug-software-maintenance.md b/papers/items/2026-2607-02606-chainswe-benchmarking-coding-agents-on-multi-bug-software-maintenance.md new file mode 100644 index 0000000..4b99c29 --- /dev/null +++ b/papers/items/2026-2607-02606-chainswe-benchmarking-coding-agents-on-multi-bug-software-maintenance.md @@ -0,0 +1,60 @@ +# Paper: ChainSWE: Benchmarking Coding Agents on Multi-Bug Software Maintenance + +--- +type: paper +title: "ChainSWE: Benchmarking Coding Agents on Multi-Bug Software Maintenance" +authors: Qirui Jin, Lingching Tung, Kenan Li, Qiyang Shi, Yushi She, Huanzhong Jia, Harrison Zhao, Kejing Xia, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02606 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, workflow-agent +- arXiv categories: cs.SE +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02606 diff --git a/papers/items/2026-2607-02684-can-coding-agents-implement-missed-compiler-optimizations-evaluating-llm-agents-.md b/papers/items/2026-2607-02684-can-coding-agents-implement-missed-compiler-optimizations-evaluating-llm-agents-.md new file mode 100644 index 0000000..1119220 --- /dev/null +++ b/papers/items/2026-2607-02684-can-coding-agents-implement-missed-compiler-optimizations-evaluating-llm-agents-.md @@ -0,0 +1,60 @@ +# Paper: Can Coding Agents Implement Missed Compiler Optimizations? Evaluating LLM Agents on LLVM Peephole Optimizations + +--- +type: paper +title: Can Coding Agents Implement Missed Compiler Optimizations? Evaluating LLM Agents on LLVM Peephole Optimizations +authors: Hongxu Xu, Chunhao Liao, Xintong Zhou, Chengnian Sun +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02684 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: coding-agent, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, llm-agent +- inferred topics: agent-evaluation, coding-agent, reasoning +- arXiv categories: cs.SE +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02684 diff --git a/papers/items/2026-2607-02689-s-ember-a-large-scale-benchmark-for-streaming-egocentric-memory-retrieval.md b/papers/items/2026-2607-02689-s-ember-a-large-scale-benchmark-for-streaming-egocentric-memory-retrieval.md new file mode 100644 index 0000000..af80040 --- /dev/null +++ b/papers/items/2026-2607-02689-s-ember-a-large-scale-benchmark-for-streaming-egocentric-memory-retrieval.md @@ -0,0 +1,63 @@ +# Paper: S-EMBER: A Large-Scale Benchmark for Streaming Egocentric Memory Retrieval + +--- +type: paper +title: "S-EMBER: A Large-Scale Benchmark for Streaming Egocentric Memory Retrieval" +authors: Xiaodong Wang, Xuanyi Zhao, Pedro Rodriguez, Devendra Singh Sachan, Barlas Oguz, Seungwhan Moon, Shang-Wen Li, Gargi Ghosh, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02689 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, memory, rag, reasoning, tool-use +- arXiv categories: cs.CV, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02689 diff --git a/papers/items/2026-2607-02703-llmoxie-exploring-agentic-ai-for-scientific-software-development.md b/papers/items/2026-2607-02703-llmoxie-exploring-agentic-ai-for-scientific-software-development.md new file mode 100644 index 0000000..24bfd41 --- /dev/null +++ b/papers/items/2026-2607-02703-llmoxie-exploring-agentic-ai-for-scientific-software-development.md @@ -0,0 +1,65 @@ +# Paper: LLMoxie: Exploring Agentic AI for Scientific Software Development + +--- +type: paper +title: "LLMoxie: Exploring Agentic AI for Scientific Software Development" +authors: Landung Setiawan, Anant Mittal, Cordero Core, Anshul Tambay, Carlos Garcia Jurado Suarez, David A. C. Beck, Andrew J. Connolly, Vani Mandava +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02703 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - planning + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.DC + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agentic-ai, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, coding-agent +- inferred topics: agent-evaluation, coding-agent, planning, reasoning, workflow-agent +- arXiv categories: cs.SE, cs.AI, cs.DC, cs.MA +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02703 diff --git a/papers/items/2026-2607-02716-evaluating-large-language-models-for-decision-making-in-agent-based-urban-mobili.md b/papers/items/2026-2607-02716-evaluating-large-language-models-for-decision-making-in-agent-based-urban-mobili.md new file mode 100644 index 0000000..deb50ad --- /dev/null +++ b/papers/items/2026-2607-02716-evaluating-large-language-models-for-decision-making-in-agent-based-urban-mobili.md @@ -0,0 +1,64 @@ +# Paper: Evaluating Large Language Models for Decision-Making in Agent-Based Urban Mobility Simulations + +--- +type: paper +title: Evaluating Large Language Models for Decision-Making in Agent-Based Urban Mobility Simulations +authors: Bruno Cascaes Alves, Míriam Blank Born, Ulisses Gilioli Francescatto Júnior, Felipe Moura Goulart, Letícia Brandão Caldas, Marilton Sanchotene de Aguiar +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02716 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - multi-agent + - planning + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, computer-use, memory, multi-agent, planning, tool-use, world-model +- arXiv categories: cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02716 diff --git a/papers/items/2026-2607-02807-swarmresearch-orchestrating-coding-agents-for-open-ended-discovery.md b/papers/items/2026-2607-02807-swarmresearch-orchestrating-coding-agents-for-open-ended-discovery.md new file mode 100644 index 0000000..a2e905e --- /dev/null +++ b/papers/items/2026-2607-02807-swarmresearch-orchestrating-coding-agents-for-open-ended-discovery.md @@ -0,0 +1,60 @@ +# Paper: SwarmResearch: Orchestrating Coding Agents for Open-Ended Discovery + +--- +type: paper +title: "SwarmResearch: Orchestrating Coding Agents for Open-Ended Discovery" +authors: Yuvraj Virk, Zack Edds, Chunqiu Steven Xia, Lingming Zhang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02807 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-02 +updated_at: 2026-07-02 +status: queued +relevance: high +topics: + - coding-agent + - computer-use + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, multi-agent-llm +- inferred topics: coding-agent, computer-use, multi-agent +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02807 diff --git a/papers/items/2026-2607-02846-object-centric-environment-modeling-for-agentic-tasks.md b/papers/items/2026-2607-02846-object-centric-environment-modeling-for-agentic-tasks.md new file mode 100644 index 0000000..3cdde19 --- /dev/null +++ b/papers/items/2026-2607-02846-object-centric-environment-modeling-for-agentic-tasks.md @@ -0,0 +1,61 @@ +# Paper: Object-Centric Environment Modeling for Agentic Tasks + +--- +type: paper +title: Object-Centric Environment Modeling for Agentic Tasks +authors: Yiyang Li, Tianyi Ma, Zehong Wang, Yijun Ma, Yanfang Ye +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02846 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, rag, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02846 diff --git a/papers/items/2026-2607-02857-mosaic-knowledge-guided-cli-command-composition-attack-in-llm-coding-agents.md b/papers/items/2026-2607-02857-mosaic-knowledge-guided-cli-command-composition-attack-in-llm-coding-agents.md new file mode 100644 index 0000000..42a6658 --- /dev/null +++ b/papers/items/2026-2607-02857-mosaic-knowledge-guided-cli-command-composition-attack-in-llm-coding-agents.md @@ -0,0 +1,63 @@ +# Paper: MOSAIC: Knowledge-Guided CLI Command Composition Attack in LLM Coding Agents + +--- +type: paper +title: "MOSAIC: Knowledge-Guided CLI Command Composition Attack in LLM Coding Agents" +authors: Jiangrong Wu, Huaijin Wang, Yihao Zhang, Yuhong Nan, Shuai Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02857 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, workflow-agent +- arXiv categories: cs.CR, cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02857 diff --git a/papers/items/2026-2607-02879-medcalc-pro-solving-complex-medical-calculations-with-llm-agents.md b/papers/items/2026-2607-02879-medcalc-pro-solving-complex-medical-calculations-with-llm-agents.md new file mode 100644 index 0000000..96d9db6 --- /dev/null +++ b/papers/items/2026-2607-02879-medcalc-pro-solving-complex-medical-calculations-with-llm-agents.md @@ -0,0 +1,59 @@ +# Paper: MedCalc-Pro: Solving Complex Medical Calculations with LLM Agents + +--- +type: paper +title: "MedCalc-Pro: Solving Complex Medical Calculations with LLM Agents" +authors: Siran Zhao, Ruihui Hou, Ziyue Huai, Chennuo Zhang, Tong Ruan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02879 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02879 diff --git a/papers/items/2026-2607-02882-diagnosis-driven-automatic-repair-for-agentic-workflow-via-symbolic-inference.md b/papers/items/2026-2607-02882-diagnosis-driven-automatic-repair-for-agentic-workflow-via-symbolic-inference.md new file mode 100644 index 0000000..c7d5551 --- /dev/null +++ b/papers/items/2026-2607-02882-diagnosis-driven-automatic-repair-for-agentic-workflow-via-symbolic-inference.md @@ -0,0 +1,61 @@ +# Paper: Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference + +--- +type: paper +title: Diagnosis-Driven Automatic Repair for Agentic Workflow via Symbolic Inference +authors: Xuyan Ma, Yawen Wang, Junjie Wang, Xiaofei Xie, Boyu Wu, Mingyang Li, Dandan Wang, Qing Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02882 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, planning, tool-use, workflow-agent +- arXiv categories: cs.SE +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02882 diff --git a/papers/items/2026-2607-02911-coact-action-preserving-observation-compression-for-coding-agents.md b/papers/items/2026-2607-02911-coact-action-preserving-observation-compression-for-coding-agents.md new file mode 100644 index 0000000..57e6907 --- /dev/null +++ b/papers/items/2026-2607-02911-coact-action-preserving-observation-compression-for-coding-agents.md @@ -0,0 +1,61 @@ +# Paper: CoACT: Action-Preserving Observation Compression for Coding Agents + +--- +type: paper +title: "CoACT: Action-Preserving Observation Compression for Coding Agents" +authors: Haorui Chen, Yuancheng Zhu, Yitong Zhang, Jia Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02911 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - coding-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: coding-agent, rag, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02911 diff --git a/papers/items/2026-2607-02927-videosearcher-empowering-video-deep-research-with-multi-tool-agentic-reasoning-v.md b/papers/items/2026-2607-02927-videosearcher-empowering-video-deep-research-with-multi-tool-agentic-reasoning-v.md new file mode 100644 index 0000000..da431f8 --- /dev/null +++ b/papers/items/2026-2607-02927-videosearcher-empowering-video-deep-research-with-multi-tool-agentic-reasoning-v.md @@ -0,0 +1,62 @@ +# Paper: VideoSearcher: Empowering Video Deep Research with Multi-Tool Agentic Reasoning via Reinforcement Learning + +--- +type: paper +title: "VideoSearcher: Empowering Video Deep Research with Multi-Tool Agentic Reasoning via Reinforcement Learning" +authors: Zhenkun Gao, Yicheng Bao, Jinlong Peng, Xueheng Li, Theo Huang, Bangwei Liu, Kunquan Li, Zhenye Gan, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02927 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, rag, reasoning, tool-use +- arXiv categories: cs.CV, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02927 diff --git a/papers/items/2026-2607-02942-a-workflow-aware-serving-layer-for-agentic-applications.md b/papers/items/2026-2607-02942-a-workflow-aware-serving-layer-for-agentic-applications.md new file mode 100644 index 0000000..1fb9946 --- /dev/null +++ b/papers/items/2026-2607-02942-a-workflow-aware-serving-layer-for-agentic-applications.md @@ -0,0 +1,61 @@ +# Paper: A Workflow-Aware Serving Layer for Agentic Applications + +--- +type: paper +title: A Workflow-Aware Serving Layer for Agentic Applications +authors: Jiayi Qian, Zishen Wan, Hanchen Yang, Chun Tao, Souvik Kundu, Tushar Krishna +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.02942 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DC + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: reasoning, tool-use, workflow-agent +- arXiv categories: cs.DC, cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.02942 diff --git a/papers/items/2026-2607-03105-orbit-q-dual-axis-benchmarking-of-autonomous-agents-in-scientific-quantum-progra.md b/papers/items/2026-2607-03105-orbit-q-dual-axis-benchmarking-of-autonomous-agents-in-scientific-quantum-progra.md new file mode 100644 index 0000000..b1f2e1d --- /dev/null +++ b/papers/items/2026-2607-03105-orbit-q-dual-axis-benchmarking-of-autonomous-agents-in-scientific-quantum-progra.md @@ -0,0 +1,60 @@ +# Paper: ORBIT-Q: Dual-axis benchmarking of autonomous agents in scientific quantum programming + +--- +type: paper +title: "ORBIT-Q: Dual-axis benchmarking of autonomous agents in scientific quantum programming" +authors: Shi-Xin Zhang, Yu-Qin Chen +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03105 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - quant-ph +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, workflow-agent +- arXiv categories: quant-ph +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03105 diff --git a/papers/items/2026-2607-03162-apeb-benchmarking-personalization-ability-of-large-language-model-agents.md b/papers/items/2026-2607-03162-apeb-benchmarking-personalization-ability-of-large-language-model-agents.md new file mode 100644 index 0000000..60458e9 --- /dev/null +++ b/papers/items/2026-2607-03162-apeb-benchmarking-personalization-ability-of-large-language-model-agents.md @@ -0,0 +1,61 @@ +# Paper: APeB: Benchmarking Personalization Ability of Large Language Model Agents + +--- +type: paper +title: "APeB: Benchmarking Personalization Ability of Large Language Model Agents" +authors: Garry Yang, Zizhe Chen, Xinru Chen, Yongqiang Chen, Jianxiang Wang, Deyu Zou, Linyi Ding, Jialiang Wu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03162 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.HC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03162 diff --git a/papers/items/2026-2607-03220-contra-red-teaming-configurations-of-personalizable-agents.md b/papers/items/2026-2607-03220-contra-red-teaming-configurations-of-personalizable-agents.md new file mode 100644 index 0000000..1ab9d25 --- /dev/null +++ b/papers/items/2026-2607-03220-contra-red-teaming-configurations-of-personalizable-agents.md @@ -0,0 +1,64 @@ +# Paper: CONTRA: Red-Teaming Configurations of Personalizable Agents + +--- +type: paper +title: "CONTRA: Red-Teaming Configurations of Personalizable Agents" +authors: Jonathan Nöther, Adish Singla, Goran Radanovic +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03220 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-evaluation, agent-safety, coding-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CR, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03220 diff --git a/papers/items/2026-2607-03233-agentic-and-generative-ai-for-open-source-intelligence-and-cyber-investigations-.md b/papers/items/2026-2607-03233-agentic-and-generative-ai-for-open-source-intelligence-and-cyber-investigations-.md new file mode 100644 index 0000000..b1f980c --- /dev/null +++ b/papers/items/2026-2607-03233-agentic-and-generative-ai-for-open-source-intelligence-and-cyber-investigations-.md @@ -0,0 +1,65 @@ +# Paper: Agentic and Generative AI for Open-Source Intelligence and Cyber Investigations: Taxonomy, Evaluation, Challenges, and Future Directions + +--- +type: paper +title: "Agentic and Generative AI for Open-Source Intelligence and Cyber Investigations: Taxonomy, Evaluation, Challenges, and Future Directions" +authors: Eduardo Almeida Palmieri, Mohamed Chahine Ghanem, Dipo Dunsin, Zubair Baig, Ed de Quincey, Kim-Kwang Raymond Choo +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03233 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.IR + - cs.SI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 22 +collection_queries: agentic-ai, rag-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, rag-agent, tool-use +- inferred topics: agent-evaluation, agent-safety, rag, reasoning, tool-use +- arXiv categories: cs.CR, cs.AI, cs.IR, cs.SI +- collection score: 22 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03233 diff --git a/papers/items/2026-2607-03269-agentic-secpbft-agentic-ai-driven-proactive-security-framework-for-wireless-pbft.md b/papers/items/2026-2607-03269-agentic-secpbft-agentic-ai-driven-proactive-security-framework-for-wireless-pbft.md new file mode 100644 index 0000000..9058a10 --- /dev/null +++ b/papers/items/2026-2607-03269-agentic-secpbft-agentic-ai-driven-proactive-security-framework-for-wireless-pbft.md @@ -0,0 +1,63 @@ +# Paper: Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks + +--- +type: paper +title: "Agentic-SecPBFT: Agentic AI-Driven Proactive Security Framework for Wireless PBFT Consensus in Mobile Ad-Hoc Networks" +authors: Haoxiang Luo, Yinqiu Liu, Ruichen Zhang, Guangyuan Liu, Gang Sun, Hongfang Yu, Zhu Han, Dong In Kim +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03269 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-safety + - computer-use + - multi-agent + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.NI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-safety, computer-use, multi-agent, rag, tool-use, world-model +- arXiv categories: cs.NI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03269 diff --git a/papers/items/2026-2607-03316-is-agentic-code-review-helpful-mining-developers-feedback-to-coderabbit-reviews-.md b/papers/items/2026-2607-03316-is-agentic-code-review-helpful-mining-developers-feedback-to-coderabbit-reviews-.md new file mode 100644 index 0000000..a413d42 --- /dev/null +++ b/papers/items/2026-2607-03316-is-agentic-code-review-helpful-mining-developers-feedback-to-coderabbit-reviews-.md @@ -0,0 +1,61 @@ +# Paper: Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild + +--- +type: paper +title: "Is Agentic Code Review Helpful? Mining Developers' Feedback to CodeRabbit Reviews in the Wild" +authors: Hong Yi Lin, Mingzhao Liang, Kla Tantithamthavorn, Patanamon Thongtanunam +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03316 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: autonomous-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm +- inferred topics: agent-safety, coding-agent, workflow-agent +- arXiv categories: cs.SE, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03316 diff --git a/papers/items/2026-2607-03333-spork-self-speculative-forking-to-accelerate-agentic-llm-inference.md b/papers/items/2026-2607-03333-spork-self-speculative-forking-to-accelerate-agentic-llm-inference.md new file mode 100644 index 0000000..2a1c5b8 --- /dev/null +++ b/papers/items/2026-2607-03333-spork-self-speculative-forking-to-accelerate-agentic-llm-inference.md @@ -0,0 +1,64 @@ +# Paper: SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference + +--- +type: paper +title: "SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference" +authors: Huajun Bai, Weiwei Lv, Huichuan Zheng, Youyou Lu, Jiwu Shu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03333 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DC + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, coding-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.DC, cs.AI, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03333 diff --git a/papers/items/2026-2607-03423-securing-multi-tool-ai-agent-chains-with-dynamic-real-time-compositional-policie.md b/papers/items/2026-2607-03423-securing-multi-tool-ai-agent-chains-with-dynamic-real-time-compositional-policie.md new file mode 100644 index 0000000..414fa93 --- /dev/null +++ b/papers/items/2026-2607-03423-securing-multi-tool-ai-agent-chains-with-dynamic-real-time-compositional-policie.md @@ -0,0 +1,62 @@ +# Paper: Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies + +--- +type: paper +title: Securing Multi-Tool AI Agent Chains With Dynamic, Real-Time Compositional Policies +authors: Chris Schneider, Kriti Faujdar, Philipp Schoenegger, Ben Bariach +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03423 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: ai-agent, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03423 diff --git a/papers/items/2026-2607-03441-no-time-like-the-present-agentic-test-time-training-for-llm-agents.md b/papers/items/2026-2607-03441-no-time-like-the-present-agentic-test-time-training-for-llm-agents.md new file mode 100644 index 0000000..94b5381 --- /dev/null +++ b/papers/items/2026-2607-03441-no-time-like-the-present-agentic-test-time-training-for-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: No Time Like the Present: Agentic Test-Time Training for LLM Agents + +--- +type: paper +title: "No Time Like the Present: Agentic Test-Time Training for LLM Agents" +authors: Yanbo Wang, Jinhua Hao, Yuze Shi, Kun Yuan, Ming Sun +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03441 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - coding-agent + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: coding-agent, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, llm-agent +- inferred topics: coding-agent, computer-use, tool-use +- arXiv categories: cs.LG, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03441 diff --git a/papers/items/2026-2607-03510-cage-1-control-assurance-and-governance-evaluation-for-enterprise-agentic-ai.md b/papers/items/2026-2607-03510-cage-1-control-assurance-and-governance-evaluation-for-enterprise-agentic-ai.md new file mode 100644 index 0000000..fe97bae --- /dev/null +++ b/papers/items/2026-2607-03510-cage-1-control-assurance-and-governance-evaluation-for-enterprise-agentic-ai.md @@ -0,0 +1,66 @@ +# Paper: CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI + +--- +type: paper +title: "CAGE-1: Control, Assurance, and Governance Evaluation for Enterprise Agentic AI" +authors: Roopam W. Sure +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03510 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - planning + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, agent-safety, memory, planning, rag, tool-use, workflow-agent +- arXiv categories: cs.SE, cs.AI, cs.CY +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03510 diff --git a/papers/items/2026-2607-03525-gameenginebench-evaluating-coding-agents-on-real-c-runtime-environments.md b/papers/items/2026-2607-03525-gameenginebench-evaluating-coding-agents-on-real-c-runtime-environments.md new file mode 100644 index 0000000..3f65e02 --- /dev/null +++ b/papers/items/2026-2607-03525-gameenginebench-evaluating-coding-agents-on-real-c-runtime-environments.md @@ -0,0 +1,64 @@ +# Paper: GameEngineBench: Evaluating Coding Agents on Real C++ Runtime Environments + +--- +type: paper +title: "GameEngineBench: Evaluating Coding Agents on Real C++ Runtime Environments" +authors: Brian La, Sejoon Chang, Ben Kim, Junyoung Bae, Aamish Ahmad Beg, Sei Chang, Gonzalo Gonzalez-Pumariega +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03525 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - embodied-agent + - memory + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, embodied-agent, memory, tool-use, world-model +- arXiv categories: cs.SE, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03525 diff --git a/papers/items/2026-2607-03601-archeval-measuring-ai-agents-as-computer-architects.md b/papers/items/2026-2607-03601-archeval-measuring-ai-agents-as-computer-architects.md new file mode 100644 index 0000000..99121c9 --- /dev/null +++ b/papers/items/2026-2607-03601-archeval-measuring-ai-agents-as-computer-architects.md @@ -0,0 +1,62 @@ +# Paper: ArchEval: Measuring AI Agents as Computer Architects + +--- +type: paper +title: "ArchEval: Measuring AI Agents as Computer Architects" +authors: Chenyu Wang, Zishen Wan, Jeffrey Ma, Shvetank Prakash, Zhenting Qi, Haebin Do, Andy Cheng, Arya Tschand, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03601 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: ai-agent, llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, llm-agent, tool-use +- inferred topics: agent-evaluation, memory, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AR +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03601 diff --git a/papers/items/2026-2607-03628-swarm-driven-multi-agent-reasoning-for-smart-city-security.md b/papers/items/2026-2607-03628-swarm-driven-multi-agent-reasoning-for-smart-city-security.md new file mode 100644 index 0000000..27b7a0f --- /dev/null +++ b/papers/items/2026-2607-03628-swarm-driven-multi-agent-reasoning-for-smart-city-security.md @@ -0,0 +1,62 @@ +# Paper: Swarm-Driven Multi-Agent Reasoning for Smart City Security + +--- +type: paper +title: Swarm-Driven Multi-Agent Reasoning for Smart City Security +authors: Saeid Jamshidi, Kawser Wazed Nafi, Carol Fung, Foutse Khomh +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03628 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-03 +updated_at: 2026-07-03 +status: queued +relevance: high +topics: + - agent-safety + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-safety, multi-agent, reasoning, tool-use +- arXiv categories: cs.CR, cs.MA +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03628 diff --git a/papers/items/2026-2607-03691-don-t-blame-the-large-language-model-how-scaffolding-evolution-shapes-coding-age.md b/papers/items/2026-2607-03691-don-t-blame-the-large-language-model-how-scaffolding-evolution-shapes-coding-age.md new file mode 100644 index 0000000..0ade46b --- /dev/null +++ b/papers/items/2026-2607-03691-don-t-blame-the-large-language-model-how-scaffolding-evolution-shapes-coding-age.md @@ -0,0 +1,63 @@ +# Paper: Don't Blame the Large Language Model: How Scaffolding Evolution Shapes Coding Agent Quality + +--- +type: paper +title: "Don't Blame the Large Language Model: How Scaffolding Evolution Shapes Coding Agent Quality" +authors: Oussama Ben Sghaier, Hao Li, Bram Adams, Ahmed E. Hassan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03691 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, reasoning, tool-use +- arXiv categories: cs.SE, cs.AI, cs.LG +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03691 diff --git a/papers/items/2026-2607-03695-social-networks-of-llm-agents.md b/papers/items/2026-2607-03695-social-networks-of-llm-agents.md new file mode 100644 index 0000000..9efec1c --- /dev/null +++ b/papers/items/2026-2607-03695-social-networks-of-llm-agents.md @@ -0,0 +1,59 @@ +# Paper: Social Networks of LLM Agents + +--- +type: paper +title: Social Networks of LLM Agents +authors: Kaixuan Liu, Guojun Xiong, Weinan Zhang, Shengpu Tang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03695 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, multi-agent +- arXiv categories: cs.LG +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03695 diff --git a/papers/items/2026-2607-03702-agent-reinforcement-learning-via-pivotal-aware-self-feedback-retry.md b/papers/items/2026-2607-03702-agent-reinforcement-learning-via-pivotal-aware-self-feedback-retry.md new file mode 100644 index 0000000..01525b3 --- /dev/null +++ b/papers/items/2026-2607-03702-agent-reinforcement-learning-via-pivotal-aware-self-feedback-retry.md @@ -0,0 +1,62 @@ +# Paper: Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry + +--- +type: paper +title: Agent Reinforcement Learning via Pivotal-Aware Self-Feedback Retry +authors: Weiyang Guo, Zesheng Shi, Longhui Zhang, Zeen Zhu, Min Zhang, Jing Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03702 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03702 diff --git a/papers/items/2026-2607-03726-selfmem-self-optimizing-memory-for-ai-agents.md b/papers/items/2026-2607-03726-selfmem-self-optimizing-memory-for-ai-agents.md new file mode 100644 index 0000000..8b91d1b --- /dev/null +++ b/papers/items/2026-2607-03726-selfmem-self-optimizing-memory-for-ai-agents.md @@ -0,0 +1,63 @@ +# Paper: SelfMem: Self-Optimizing Memory for AI Agents + +--- +type: paper +title: "SelfMem: Self-Optimizing Memory for AI Agents" +authors: Shu Yang, Junchao Wu, Derek F. Wong, Di Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03726 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-memory, ai-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, ai-agent, tool-use +- inferred topics: agent-evaluation, computer-use, memory, planning, rag, tool-use +- arXiv categories: cs.CL +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03726 diff --git a/papers/items/2026-2607-03821-dualview-preventing-indirect-prompt-injection-in-personal-ai-agents.md b/papers/items/2026-2607-03821-dualview-preventing-indirect-prompt-injection-in-personal-ai-agents.md new file mode 100644 index 0000000..6b82d24 --- /dev/null +++ b/papers/items/2026-2607-03821-dualview-preventing-indirect-prompt-injection-in-personal-ai-agents.md @@ -0,0 +1,61 @@ +# Paper: DualView: Preventing Indirect Prompt Injection in Personal AI Agents + +--- +type: paper +title: "DualView: Preventing Indirect Prompt Injection in Personal AI Agents" +authors: Juhee Kim, Woohyuk Choi, Taehyun Kang, Youngmin Kim, Byoungyoung Lee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03821 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03821 diff --git a/papers/items/2026-2607-03853-cograd-a-cognitively-inspired-multi-agent-framework-for-radiology-report-generat.md b/papers/items/2026-2607-03853-cograd-a-cognitively-inspired-multi-agent-framework-for-radiology-report-generat.md new file mode 100644 index 0000000..6af619a --- /dev/null +++ b/papers/items/2026-2607-03853-cograd-a-cognitively-inspired-multi-agent-framework-for-radiology-report-generat.md @@ -0,0 +1,61 @@ +# Paper: CogRad: A Cognitively-Inspired Multi-Agent Framework for Radiology Report Generation + +--- +type: paper +title: "CogRad: A Cognitively-Inspired Multi-Agent Framework for Radiology Report Generation" +authors: Saif Ur Rehman Khan, Hasaan Maqsood, Sebastian Vollmer, Andreas Dengel, Muhammad Nabeel Asim +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03853 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, rag, reasoning +- arXiv categories: cs.CV +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03853 diff --git a/papers/items/2026-2607-03953-the-remarkable-effectiveness-of-providing-ai-agents-with-natural-language-tools-.md b/papers/items/2026-2607-03953-the-remarkable-effectiveness-of-providing-ai-agents-with-natural-language-tools-.md new file mode 100644 index 0000000..8dd32f2 --- /dev/null +++ b/papers/items/2026-2607-03953-the-remarkable-effectiveness-of-providing-ai-agents-with-natural-language-tools-.md @@ -0,0 +1,63 @@ +# Paper: The Remarkable Effectiveness of Providing AI Agents with Natural Language Tools: A Replication Study Validating NLT Performance Across 14 Models + +--- +type: paper +title: "The Remarkable Effectiveness of Providing AI Agents with Natural Language Tools: A Replication Study Validating NLT Performance Across 14 Models" +authors: Alexander Somma, Isabelle Plante, Fred Premji +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03953 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agentic-ai, ai-agent, llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, ai-agent, llm-agent, tool-use +- inferred topics: agent-evaluation, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.CL, cs.AI +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03953 diff --git a/papers/items/2026-2607-03968-refused-in-chat-written-in-code-workflow-level-jailbreak-construction-in-ide-cod.md b/papers/items/2026-2607-03968-refused-in-chat-written-in-code-workflow-level-jailbreak-construction-in-ide-cod.md new file mode 100644 index 0000000..655a6df --- /dev/null +++ b/papers/items/2026-2607-03968-refused-in-chat-written-in-code-workflow-level-jailbreak-construction-in-ide-cod.md @@ -0,0 +1,62 @@ +# Paper: Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents + +--- +type: paper +title: "Refused in Chat, Written in Code: Workflow-Level Jailbreak Construction in IDE Coding Agents" +authors: Abhishek Kumar, Carsten Maple +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.03968 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, workflow-agent +- arXiv categories: cs.SE, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.03968 diff --git a/papers/items/2026-2607-04009-physminer-an-agentic-ai-framework-for-discovering-turbulence-physics.md b/papers/items/2026-2607-04009-physminer-an-agentic-ai-framework-for-discovering-turbulence-physics.md new file mode 100644 index 0000000..cbd02f8 --- /dev/null +++ b/papers/items/2026-2607-04009-physminer-an-agentic-ai-framework-for-discovering-turbulence-physics.md @@ -0,0 +1,61 @@ +# Paper: PhysMiner: An Agentic AI Framework for Discovering Turbulence Physics + +--- +type: paper +title: "PhysMiner: An Agentic AI Framework for Discovering Turbulence Physics" +authors: Jiawei Chen, Han Gao, Ping He +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04009 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - physics.flu-dyn +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, memory, reasoning, tool-use +- arXiv categories: physics.flu-dyn +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04009 diff --git a/papers/items/2026-2607-04034-the-i-don-t-know-filter-enhancing-agentic-reliability-in-function-calling.md b/papers/items/2026-2607-04034-the-i-don-t-know-filter-enhancing-agentic-reliability-in-function-calling.md new file mode 100644 index 0000000..d80a6e7 --- /dev/null +++ b/papers/items/2026-2607-04034-the-i-don-t-know-filter-enhancing-agentic-reliability-in-function-calling.md @@ -0,0 +1,61 @@ +# Paper: The "I Don't Know" Filter: Enhancing Agentic Reliability in Function Calling + +--- +type: paper +title: "The \"I Don't Know\" Filter: Enhancing Agentic Reliability in Function Calling" +authors: Stefan Broecker, Mason del Rosario, Boris Selitser, Thomas Strohmer +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04034 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-04 +updated_at: 2026-07-04 +status: queued +relevance: high +topics: + - agent-evaluation + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation, function-calling +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, function-calling +- inferred topics: agent-evaluation, rag, tool-use +- arXiv categories: cs.SE, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04034 diff --git a/papers/items/2026-2607-04089-placemem-toward-a-compute-aware-memory-plane-for-lifelong-agents.md b/papers/items/2026-2607-04089-placemem-toward-a-compute-aware-memory-plane-for-lifelong-agents.md new file mode 100644 index 0000000..cffa3a0 --- /dev/null +++ b/papers/items/2026-2607-04089-placemem-toward-a-compute-aware-memory-plane-for-lifelong-agents.md @@ -0,0 +1,61 @@ +# Paper: PLACEMEM: Toward a Compute-Aware Memory Plane for Lifelong Agents + +--- +type: paper +title: "PLACEMEM: Toward a Compute-Aware Memory Plane for Lifelong Agents" +authors: Sukanta Ganguly +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04089 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agent-memory +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory +- inferred topics: agent-evaluation, memory, planning, rag +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04089 diff --git a/papers/items/2026-2607-04149-beyond-scene-priors-fine-grained-traffic-scene-reasoning-with-benchmarking-and-q.md b/papers/items/2026-2607-04149-beyond-scene-priors-fine-grained-traffic-scene-reasoning-with-benchmarking-and-q.md new file mode 100644 index 0000000..8751728 --- /dev/null +++ b/papers/items/2026-2607-04149-beyond-scene-priors-fine-grained-traffic-scene-reasoning-with-benchmarking-and-q.md @@ -0,0 +1,63 @@ +# Paper: Beyond Scene Priors: Fine-Grained Traffic Scene Reasoning with Benchmarking and Query-Guided Small-Object Focus + +--- +type: paper +title: "Beyond Scene Priors: Fine-Grained Traffic Scene Reasoning with Benchmarking and Query-Guided Small-Object Focus" +authors: Waikit Xiu, Qiang Lu, Zian Wang, Xinjie Yang, Zhiwei Chen, Chen Sun, Xiying Li +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04149 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - multi-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, multi-agent, reasoning +- arXiv categories: cs.CV +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04149 diff --git a/papers/items/2026-2607-04162-ace-agentic-control-for-embodied-manipulation-via-zero-shot-workflow-reasoning.md b/papers/items/2026-2607-04162-ace-agentic-control-for-embodied-manipulation-via-zero-shot-workflow-reasoning.md new file mode 100644 index 0000000..4217453 --- /dev/null +++ b/papers/items/2026-2607-04162-ace-agentic-control-for-embodied-manipulation-via-zero-shot-workflow-reasoning.md @@ -0,0 +1,66 @@ +# Paper: ACE: Agentic Control for Embodied Manipulation via Zero-shot Workflow Reasoning + +--- +type: paper +title: "ACE: Agentic Control for Embodied Manipulation via Zero-shot Workflow Reasoning" +authors: Iok Tong Lei, QianZhi Li, Ying Jie Yap, Yujie Zhang, Rui Zhong, Haichao Gui, Xiaolong Liu, Zhidong Deng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04162 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, embodied-agent, memory, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.RO, cs.LG +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04162 diff --git a/papers/items/2026-2607-04212-an-evaluation-of-role-based-multi-agent-code-generation-on-repository-scale-prob.md b/papers/items/2026-2607-04212-an-evaluation-of-role-based-multi-agent-code-generation-on-repository-scale-prob.md new file mode 100644 index 0000000..251e323 --- /dev/null +++ b/papers/items/2026-2607-04212-an-evaluation-of-role-based-multi-agent-code-generation-on-repository-scale-prob.md @@ -0,0 +1,60 @@ +# Paper: An Evaluation of Role-Based Multi-Agent Code Generation on Repository-Scale Problems + +--- +type: paper +title: An Evaluation of Role-Based Multi-Agent Code Generation on Repository-Scale Problems +authors: Benedetta Donato, Noah Hagar-Dent, Aaron Worsnop, Leonardo Mariani, Valerio Terragni +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04212 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent +- arXiv categories: cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04212 diff --git a/papers/items/2026-2607-04219-agentic-iot-architectures-applications-and-challenges-toward-the-internet-of-age.md b/papers/items/2026-2607-04219-agentic-iot-architectures-applications-and-challenges-toward-the-internet-of-age.md new file mode 100644 index 0000000..95999eb --- /dev/null +++ b/papers/items/2026-2607-04219-agentic-iot-architectures-applications-and-challenges-toward-the-internet-of-age.md @@ -0,0 +1,63 @@ +# Paper: Agentic IoT: Architectures, Applications, and Challenges Toward the Internet of Agents + +--- +type: paper +title: "Agentic IoT: Architectures, Applications, and Challenges Toward the Internet of Agents" +authors: Rümeysa Hilal Sevinç, Bahaeddin Türkoğlu, İbrahim Kök +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04219 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA + - cs.NI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: ai-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, tool-use +- inferred topics: multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.AI, cs.MA, cs.NI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04219 diff --git a/papers/items/2026-2607-04240-biological-motifs-for-agentic-control.md b/papers/items/2026-2607-04240-biological-motifs-for-agentic-control.md new file mode 100644 index 0000000..4e675b4 --- /dev/null +++ b/papers/items/2026-2607-04240-biological-motifs-for-agentic-control.md @@ -0,0 +1,62 @@ +# Paper: Biological Motifs for Agentic Control + +--- +type: paper +title: Biological Motifs for Agentic Control +authors: Bogdan Banu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04240 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - multi-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - q-bio.CB +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agent-evaluation, autonomous-agent-llm, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, autonomous-agent-llm, multi-agent-llm +- inferred topics: agent-evaluation, agent-safety, multi-agent, tool-use +- arXiv categories: cs.AI, q-bio.CB +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04240 diff --git a/papers/items/2026-2607-04293-causalgame-benchmarking-causal-thinking-of-llm-agents-in-games.md b/papers/items/2026-2607-04293-causalgame-benchmarking-causal-thinking-of-llm-agents-in-games.md new file mode 100644 index 0000000..36677fe --- /dev/null +++ b/papers/items/2026-2607-04293-causalgame-benchmarking-causal-thinking-of-llm-agents-in-games.md @@ -0,0 +1,64 @@ +# Paper: CausalGame: Benchmarking Causal Thinking of LLM Agents in Games + +--- +type: paper +title: "CausalGame: Benchmarking Causal Thinking of LLM Agents in Games" +authors: Zhenhao Chen, Yongqiang Chen, Chenxi Liu, Junchi Yu, Xiangchen Song, Zijian Li, Jialin Li, Philip Torr, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04293 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.LG + - stat.ML +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, computer-use, planning, reasoning +- arXiv categories: cs.CL, cs.AI, cs.LG, stat.ML +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04293 diff --git a/papers/items/2026-2607-04334-do-gui-agents-believe-their-eyes-diagnosing-state-belief-reliance-on-pixels-vers.md b/papers/items/2026-2607-04334-do-gui-agents-believe-their-eyes-diagnosing-state-belief-reliance-on-pixels-vers.md new file mode 100644 index 0000000..604c506 --- /dev/null +++ b/papers/items/2026-2607-04334-do-gui-agents-believe-their-eyes-diagnosing-state-belief-reliance-on-pixels-vers.md @@ -0,0 +1,61 @@ +# Paper: Do GUI Agents Believe Their Eyes? Diagnosing State-Belief Reliance on Pixels versus Structure + +--- +type: paper +title: Do GUI Agents Believe Their Eyes? Diagnosing State-Belief Reliance on Pixels versus Structure +authors: Guijia Zhang, Harry Yang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04334 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: web-gui-agent +- inferred topics: agent-evaluation, computer-use, rag, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04334 diff --git a/papers/items/2026-2607-04391-memory-orchestrated-semantic-system-moss-an-auditable-agentic-memory-architectur.md b/papers/items/2026-2607-04391-memory-orchestrated-semantic-system-moss-an-auditable-agentic-memory-architectur.md new file mode 100644 index 0000000..22b35da --- /dev/null +++ b/papers/items/2026-2607-04391-memory-orchestrated-semantic-system-moss-an-auditable-agentic-memory-architectur.md @@ -0,0 +1,61 @@ +# Paper: Memory-Orchestrated Semantic System (MOSS): An Auditable Agentic Memory Architecture + +--- +type: paper +title: "Memory-Orchestrated Semantic System (MOSS): An Auditable Agentic Memory Architecture" +authors: Serge Lacasse, Jérémie Hatier, Alex Baker +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04391 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: agent-memory, ai-agent, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, ai-agent, rag-agent +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.CL +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04391 diff --git a/papers/items/2026-2607-04394-mechmath-agent-team-llm-driven-agents-for-mathematical-research.md b/papers/items/2026-2607-04394-mechmath-agent-team-llm-driven-agents-for-mathematical-research.md new file mode 100644 index 0000000..cc48000 --- /dev/null +++ b/papers/items/2026-2607-04394-mechmath-agent-team-llm-driven-agents-for-mathematical-research.md @@ -0,0 +1,63 @@ +# Paper: MechMath Agent Team: LLM Driven Agents for Mathematical Research + +--- +type: paper +title: "MechMath Agent Team: LLM Driven Agents for Mathematical Research" +authors: Yichuan Cao, Ruichen Qiu, Junqi Liu, Jiaqi Wang, Dakai Guo, Ruyong Feng, Lihong Zhi, Xiao-Shan Gao +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04394 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - planning + - rag + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.SC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, planning, rag, reasoning +- arXiv categories: cs.AI, cs.SC +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04394 diff --git a/papers/items/2026-2607-04395-nki-agent-domain-specific-fine-tuning-and-agentic-tool-use-for-neuron-kernel-gen.md b/papers/items/2026-2607-04395-nki-agent-domain-specific-fine-tuning-and-agentic-tool-use-for-neuron-kernel-gen.md new file mode 100644 index 0000000..062cc4b --- /dev/null +++ b/papers/items/2026-2607-04395-nki-agent-domain-specific-fine-tuning-and-agentic-tool-use-for-neuron-kernel-gen.md @@ -0,0 +1,61 @@ +# Paper: NKI-Agent: Domain-Specific Fine-Tuning and Agentic Tool Use for Neuron Kernel Generation + +--- +type: paper +title: "NKI-Agent: Domain-Specific Fine-Tuning and Agentic Tool Use for Neuron Kernel Generation" +authors: Junjie Tang, Jun Huan, Hao Zhou, Yuhao Zhang, Lin Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04395 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, computer-use, memory, tool-use +- arXiv categories: cs.LG +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04395 diff --git a/papers/items/2026-2607-04426-ace-brain-0-5-a-unified-embodied-foundational-model-for-physical-agentic-ai.md b/papers/items/2026-2607-04426-ace-brain-0-5-a-unified-embodied-foundational-model-for-physical-agentic-ai.md new file mode 100644 index 0000000..2acd0ca --- /dev/null +++ b/papers/items/2026-2607-04426-ace-brain-0-5-a-unified-embodied-foundational-model-for-physical-agentic-ai.md @@ -0,0 +1,64 @@ +# Paper: ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI + +--- +type: paper +title: "ACE-Brain-0.5: A Unified Embodied Foundational Model for Physical Agentic AI" +authors: "ACE-Brain Team, :, Ziyang Gong, Haoming Gu, Zehang Luo, Tianyi Zhang, Tao Tao, Yixiao Chi, et al." +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04426 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agentic-ai +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai +- inferred topics: agent-evaluation, embodied-agent, memory, planning, rag, reasoning, tool-use +- arXiv categories: cs.RO +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04426 diff --git a/papers/items/2026-2607-04433-autonomous-information-seeking-a-roadmap-for-agentic-recommender-systems.md b/papers/items/2026-2607-04433-autonomous-information-seeking-a-roadmap-for-agentic-recommender-systems.md new file mode 100644 index 0000000..d3a99c9 --- /dev/null +++ b/papers/items/2026-2607-04433-autonomous-information-seeking-a-roadmap-for-agentic-recommender-systems.md @@ -0,0 +1,66 @@ +# Paper: Autonomous Information Seeking: A Roadmap for Agentic Recommender Systems + +--- +type: paper +title: "Autonomous Information Seeking: A Roadmap for Agentic Recommender Systems" +authors: Xinyu Lin, Yashar Deldjoo, Sunhao Dai, Honghui Bao, Xiaopeng Ye, Fatemeh Nazary, Wenjie Wang, Tommaso Di Noia, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04433 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - planning + - reasoning + - tool-use + - workflow-agent + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.IR + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, agent-safety, memory, planning, reasoning, tool-use, workflow-agent, world-model +- arXiv categories: cs.IR, cs.CL +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04433 diff --git a/papers/items/2026-2607-04470-regime-conditional-stabilisation-of-llm-augmented-cooperative-multi-agent-reinfo.md b/papers/items/2026-2607-04470-regime-conditional-stabilisation-of-llm-augmented-cooperative-multi-agent-reinfo.md new file mode 100644 index 0000000..e0477da --- /dev/null +++ b/papers/items/2026-2607-04470-regime-conditional-stabilisation-of-llm-augmented-cooperative-multi-agent-reinfo.md @@ -0,0 +1,63 @@ +# Paper: Regime-Conditional Stabilisation of LLM-Augmented Cooperative Multi-Agent Reinforcement Learning + +--- +type: paper +title: Regime-Conditional Stabilisation of LLM-Augmented Cooperative Multi-Agent Reinforcement Learning +authors: Faid Keddouri, Sohaib Houhou, Aissa Boulmerka, Nadir Farhi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04470 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI + - math.OC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, rag, tool-use +- arXiv categories: cs.LG, cs.AI, math.OC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04470 diff --git a/papers/items/2026-2607-04528-measuring-harness-induced-belief-divergence-in-multi-step-llm-agents.md b/papers/items/2026-2607-04528-measuring-harness-induced-belief-divergence-in-multi-step-llm-agents.md new file mode 100644 index 0000000..58cf0a6 --- /dev/null +++ b/papers/items/2026-2607-04528-measuring-harness-induced-belief-divergence-in-multi-step-llm-agents.md @@ -0,0 +1,61 @@ +# Paper: Measuring Harness-Induced Belief Divergence in Multi-Step LLM Agents + +--- +type: paper +title: Measuring Harness-Induced Belief Divergence in Multi-Step LLM Agents +authors: Haiwen Yi, Xinyuan Song +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04528 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, llm-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04528 diff --git a/papers/items/2026-2607-04569-llms-for-agentic-home-energy-management.md b/papers/items/2026-2607-04569-llms-for-agentic-home-energy-management.md new file mode 100644 index 0000000..d962db0 --- /dev/null +++ b/papers/items/2026-2607-04569-llms-for-agentic-home-energy-management.md @@ -0,0 +1,61 @@ +# Paper: LLMs for Agentic Home Energy Management + +--- +type: paper +title: LLMs for Agentic Home Energy Management +authors: Sokipriala Jonah +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04569 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - eess.SY +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: function-calling, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling, llm-agent +- inferred topics: agent-evaluation, agent-safety, rag, tool-use +- arXiv categories: eess.SY +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04569 diff --git a/papers/items/2026-2607-04617-mrms-a-multi-resolution-memory-substrate-for-long-lived-ai-agents.md b/papers/items/2026-2607-04617-mrms-a-multi-resolution-memory-substrate-for-long-lived-ai-agents.md new file mode 100644 index 0000000..cdc2b13 --- /dev/null +++ b/papers/items/2026-2607-04617-mrms-a-multi-resolution-memory-substrate-for-long-lived-ai-agents.md @@ -0,0 +1,62 @@ +# Paper: MRMS: A Multi-Resolution Memory Substrate for Long-Lived AI Agents + +--- +type: paper +title: "MRMS: A Multi-Resolution Memory Substrate for Long-Lived AI Agents" +authors: Jizhizi Li, Amy Shi-Nash +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04617 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, computer-use, memory, rag, tool-use +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04617 diff --git a/papers/items/2026-2607-04623-can-llms-really-recover-microservice-failures-a-recovery-aware-evaluation-of-dia.md b/papers/items/2026-2607-04623-can-llms-really-recover-microservice-failures-a-recovery-aware-evaluation-of-dia.md new file mode 100644 index 0000000..40c9bd0 --- /dev/null +++ b/papers/items/2026-2607-04623-can-llms-really-recover-microservice-failures-a-recovery-aware-evaluation-of-dia.md @@ -0,0 +1,63 @@ +# Paper: Can LLMs Really Recover Microservice Failures? A Recovery-Aware Evaluation of Diagnosis-to-Action Reasoning + +--- +type: paper +title: Can LLMs Really Recover Microservice Failures? A Recovery-Aware Evaluation of Diagnosis-to-Action Reasoning +authors: Jiaxing Qi, Zhongzhi Luan, Hongyu Zhang, Shaohan Huang, Carol Fung, Yongxin Tong, Hailong Yang, Depei Qian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04623 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.DC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, planning, rag, reasoning, tool-use +- arXiv categories: cs.SE, cs.DC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04623 diff --git a/papers/items/2026-2607-04686-toolfailbench-diagnosing-tool-use-failures-in-llm-agents.md b/papers/items/2026-2607-04686-toolfailbench-diagnosing-tool-use-failures-in-llm-agents.md new file mode 100644 index 0000000..222f6c1 --- /dev/null +++ b/papers/items/2026-2607-04686-toolfailbench-diagnosing-tool-use-failures-in-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: ToolFailBench: Diagnosing Tool-Use Failures in LLM Agents + +--- +type: paper +title: "ToolFailBench: Diagnosing Tool-Use Failures in LLM Agents" +authors: Harsh Soni +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04686 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: agentic-ai, llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, llm-agent, tool-use +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CL, cs.AI, cs.SE +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04686 diff --git a/papers/items/2026-2607-04697-ai-agent-pull-requests-on-github-frequency-structure-and-merge-conflict-rates.md b/papers/items/2026-2607-04697-ai-agent-pull-requests-on-github-frequency-structure-and-merge-conflict-rates.md new file mode 100644 index 0000000..282983d --- /dev/null +++ b/papers/items/2026-2607-04697-ai-agent-pull-requests-on-github-frequency-structure-and-merge-conflict-rates.md @@ -0,0 +1,60 @@ +# Paper: AI Agent Pull Requests on GitHub: Frequency, Structure, and Merge Conflict Rates + +--- +type: paper +title: "AI Agent Pull Requests on GitHub: Frequency, Structure, and Merge Conflict Rates" +authors: George Xu, Arjun Subramanian, Nithilan Karthik +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04697 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, coding-agent +- inferred topics: agent-evaluation, coding-agent, multi-agent +- arXiv categories: cs.SE +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04697 diff --git a/papers/items/2026-2607-04713-rspo-reward-swap-policy-optimization-for-multi-turn-llm-agents.md b/papers/items/2026-2607-04713-rspo-reward-swap-policy-optimization-for-multi-turn-llm-agents.md new file mode 100644 index 0000000..4587f2e --- /dev/null +++ b/papers/items/2026-2607-04713-rspo-reward-swap-policy-optimization-for-multi-turn-llm-agents.md @@ -0,0 +1,62 @@ +# Paper: RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents + +--- +type: paper +title: "RSPO: Reward-Swap Policy Optimization for Multi-Turn LLM Agents" +authors: Qiang Liu, Taian Guo, Ruizhi Qiao, Xing Sun +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04713 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, llm-agent +- inferred topics: agent-evaluation, agent-safety, planning, rag +- arXiv categories: cs.LG, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04713 diff --git a/papers/items/2026-2607-04963-stapo-selective-trajectory-aware-policy-optimization-for-llm-agent-training.md b/papers/items/2026-2607-04963-stapo-selective-trajectory-aware-policy-optimization-for-llm-agent-training.md new file mode 100644 index 0000000..0e084fb --- /dev/null +++ b/papers/items/2026-2607-04963-stapo-selective-trajectory-aware-policy-optimization-for-llm-agent-training.md @@ -0,0 +1,60 @@ +# Paper: STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training + +--- +type: paper +title: "STAPO: Selective Trajectory-Aware Policy Optimization for LLM Agent Training" +authors: Qiuyi Qi, Tian Liang, Mutian Bao, Jinjian Zhang, Dongnan Liu, Wei Zhou, Linjian Mo, Ming Kong, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.04963 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - planning + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: planning, rag, tool-use +- arXiv categories: cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.04963 diff --git a/papers/items/2026-2607-05001-tactic-kg-toward-small-agent-teams-for-cyber-threat-intelligence-knowledge-graph.md b/papers/items/2026-2607-05001-tactic-kg-toward-small-agent-teams-for-cyber-threat-intelligence-knowledge-graph.md new file mode 100644 index 0000000..5e8bc27 --- /dev/null +++ b/papers/items/2026-2607-05001-tactic-kg-toward-small-agent-teams-for-cyber-threat-intelligence-knowledge-graph.md @@ -0,0 +1,64 @@ +# Paper: TACTIC-KG: Toward Small Agent Teams for Cyber Threat Intelligence Knowledge Graph Construction + +--- +type: paper +title: "TACTIC-KG: Toward Small Agent Teams for Cyber Threat Intelligence Knowledge Graph Construction" +authors: Mouhamed Amine Bouchiha, Gregory Blanc +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05001 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI + - cs.LG + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, agent-safety, reasoning, tool-use +- arXiv categories: cs.CR, cs.AI, cs.LG, cs.MA +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05001 diff --git a/papers/items/2026-2607-05029-your-agent-s-memories-are-not-its-own-forged-reasoning-attacks-on-llm-agent-memo.md b/papers/items/2026-2607-05029-your-agent-s-memories-are-not-its-own-forged-reasoning-attacks-on-llm-agent-memo.md new file mode 100644 index 0000000..8bae70b --- /dev/null +++ b/papers/items/2026-2607-05029-your-agent-s-memories-are-not-its-own-forged-reasoning-attacks-on-llm-agent-memo.md @@ -0,0 +1,62 @@ +# Paper: Your Agent's Memories Are Not Its Own: Forged Reasoning Attacks on LLM Agent Memory and Defenses + +--- +type: paper +title: "Your Agent's Memories Are Not Its Own: Forged Reasoning Attacks on LLM Agent Memory and Defenses" +authors: Neeraj Karamchandani, Piyush Nagasubramaniam, Sencun Zhu, Dinghao Wu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05029 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-memory, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-memory, llm-agent +- inferred topics: agent-evaluation, memory, reasoning, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05029 diff --git a/papers/items/2026-2607-05055-toward-trustworthy-large-language-model-agents-in-healthcare.md b/papers/items/2026-2607-05055-toward-trustworthy-large-language-model-agents-in-healthcare.md new file mode 100644 index 0000000..e862959 --- /dev/null +++ b/papers/items/2026-2607-05055-toward-trustworthy-large-language-model-agents-in-healthcare.md @@ -0,0 +1,62 @@ +# Paper: Toward Trustworthy Large Language Model Agents in Healthcare + +--- +type: paper +title: Toward Trustworthy Large Language Model Agents in Healthcare +authors: Hadi Hasan, Safaa Salman, Adam Tai Abou Dargham, Ammar Mohanna, Ali Chehab +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05055 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: function-calling, rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: function-calling, rag-agent +- inferred topics: agent-evaluation, agent-safety, rag, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05055 diff --git a/papers/items/2026-2607-05120-agent-data-injection-attacks-are-realistic-threats-to-ai-agents.md b/papers/items/2026-2607-05120-agent-data-injection-attacks-are-realistic-threats-to-ai-agents.md new file mode 100644 index 0000000..b6809c0 --- /dev/null +++ b/papers/items/2026-2607-05120-agent-data-injection-attacks-are-realistic-threats-to-ai-agents.md @@ -0,0 +1,63 @@ +# Paper: Agent Data Injection Attacks are Realistic Threats to AI Agents + +--- +type: paper +title: Agent Data Injection Attacks are Realistic Threats to AI Agents +authors: Woohyuk Choi, Juhee Kim, Taehyun Kang, Jihyeon Jeong, Luyi Xing, Byoungyoung Lee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05120 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - computer-use + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-safety, ai-agent, coding-agent, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-safety, ai-agent, coding-agent, web-gui-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, computer-use, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05120 diff --git a/papers/items/2026-2607-05132-when-agents-lie-premeditation-persistence-and-exploitation-in-repeated-games.md b/papers/items/2026-2607-05132-when-agents-lie-premeditation-persistence-and-exploitation-in-repeated-games.md new file mode 100644 index 0000000..f2766d5 --- /dev/null +++ b/papers/items/2026-2607-05132-when-agents-lie-premeditation-persistence-and-exploitation-in-repeated-games.md @@ -0,0 +1,62 @@ +# Paper: When Agents Lie: Premeditation, Persistence, and Exploitation in Repeated Games + +--- +type: paper +title: "When Agents Lie: Premeditation, Persistence, and Exploitation in Repeated Games" +authors: Jerick Shi, Terry Jingcheng Zhang, Bernhard Schölkopf, Vincent Conitzer, Zhijing Jin +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05132 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CY + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: autonomous-agent-llm, llm-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, llm-agent, planning-agent +- inferred topics: agent-evaluation, agent-safety, planning, tool-use +- arXiv categories: cs.CY, cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05132 diff --git a/papers/items/2026-2607-05174-agentgym2-benchmarking-large-language-model-agents-in-de-idealized-real-world-en.md b/papers/items/2026-2607-05174-agentgym2-benchmarking-large-language-model-agents-in-de-idealized-real-world-en.md new file mode 100644 index 0000000..78b10e4 --- /dev/null +++ b/papers/items/2026-2607-05174-agentgym2-benchmarking-large-language-model-agents-in-de-idealized-real-world-en.md @@ -0,0 +1,61 @@ +# Paper: AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments + +--- +type: paper +title: "AgentGym2: Benchmarking Large Language Model Agents in De-Idealized Real-World Environments" +authors: Zhiheng Xi, Dingwen Yang, Jiaqi Liu, Jixuan Huang, Honglin Guo, Baodai Huang, Tinggang Chen, Qi Zhang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05174 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: language-agent, llm-agent, planning-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent, llm-agent, planning-agent +- inferred topics: agent-evaluation, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05174 diff --git a/papers/items/2026-2607-05188-latent-programming-horizons-in-coding-agents.md b/papers/items/2026-2607-05188-latent-programming-horizons-in-coding-agents.md new file mode 100644 index 0000000..a6f9128 --- /dev/null +++ b/papers/items/2026-2607-05188-latent-programming-horizons-in-coding-agents.md @@ -0,0 +1,61 @@ +# Paper: Latent Programming Horizons in Coding Agents + +--- +type: paper +title: Latent Programming Horizons in Coding Agents +authors: André Silva, Han Tu, Martin Monperrus +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05188 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, reasoning +- arXiv categories: cs.LG, cs.SE +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05188 diff --git a/papers/items/2026-2607-05202-evoagentbench-benchmarking-agent-self-evolution-via-ability-transfer.md b/papers/items/2026-2607-05202-evoagentbench-benchmarking-agent-self-evolution-via-ability-transfer.md new file mode 100644 index 0000000..8aa9b68 --- /dev/null +++ b/papers/items/2026-2607-05202-evoagentbench-benchmarking-agent-self-evolution-via-ability-transfer.md @@ -0,0 +1,63 @@ +# Paper: EvoAgentBench: Benchmarking Agent Self-Evolution via Ability Transfer + +--- +type: paper +title: "EvoAgentBench: Benchmarking Agent Self-Evolution via Ability Transfer" +authors: Xingze Gao, Chuanrui Hu, Hongda Chen, Pengfei Yao, Zhao Wang, Yi Bai, Zhengwei Wu, Yunyun Han, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05202 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - memory + - planning + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 19 +collection_queries: agent-evaluation +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation +- inferred topics: agent-evaluation, coding-agent, computer-use, memory, planning, reasoning +- arXiv categories: cs.AI +- collection score: 19 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05202 diff --git a/papers/items/2026-2607-05297-metaskill-evolve-recursive-self-improvement-of-llm-agents-via-two-timescale-meta.md b/papers/items/2026-2607-05297-metaskill-evolve-recursive-self-improvement-of-llm-agents-via-two-timescale-meta.md new file mode 100644 index 0000000..49f4be4 --- /dev/null +++ b/papers/items/2026-2607-05297-metaskill-evolve-recursive-self-improvement-of-llm-agents-via-two-timescale-meta.md @@ -0,0 +1,61 @@ +# Paper: MetaSkill-Evolve: Recursive Self-Improvement of LLM Agents via Two-Timescale Meta-Skill Evolution + +--- +type: paper +title: "MetaSkill-Evolve: Recursive Self-Improvement of LLM Agents via Two-Timescale Meta-Skill Evolution" +authors: Zefeng Wang, Minxi Yan, Jinhe Bi, Sikuan Yan, Volker Tresp, Yunpu Ma +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05297 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - planning + - reasoning + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: agent-evaluation, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, llm-agent +- inferred topics: agent-evaluation, planning, reasoning, workflow-agent +- arXiv categories: cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05297 diff --git a/papers/items/2026-2607-05318-pisas-benchmarking-contextual-integrity-in-multi-user-agentic-systems.md b/papers/items/2026-2607-05318-pisas-benchmarking-contextual-integrity-in-multi-user-agentic-systems.md new file mode 100644 index 0000000..71ecade --- /dev/null +++ b/papers/items/2026-2607-05318-pisas-benchmarking-contextual-integrity-in-multi-user-agentic-systems.md @@ -0,0 +1,62 @@ +# Paper: PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems + +--- +type: paper +title: "PiSAs: Benchmarking Contextual Integrity in Multi-User Agentic Systems" +authors: Shubham Gupta, Nazanin Mohammadi Sepahvand, Abhinav Kumar, Cem Subakan, Spandana Gella, Pierre-André Noël, Perouz Taslakian, Eugene Bagdasarian, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05318 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.MA + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 20 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, agent-safety, memory, tool-use +- arXiv categories: cs.MA, cs.CR +- collection score: 20 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05318 diff --git a/papers/items/2026-2607-05363-sovereignpa-bench-evaluating-user-owned-personal-agents-under-evolving-intent-pl.md b/papers/items/2026-2607-05363-sovereignpa-bench-evaluating-user-owned-personal-agents-under-evolving-intent-pl.md new file mode 100644 index 0000000..8c637d3 --- /dev/null +++ b/papers/items/2026-2607-05363-sovereignpa-bench-evaluating-user-owned-personal-agents-under-evolving-intent-pl.md @@ -0,0 +1,63 @@ +# Paper: SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints + +--- +type: paper +title: "SovereignPA-Bench: Evaluating User-Owned Personal Agents under Evolving Intent, Platform Mediation, and Consent Constraints" +authors: Dylan Zongmin Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05363 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - embodied-agent + - memory + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, tool-use +- inferred topics: agent-evaluation, agent-safety, computer-use, embodied-agent, memory, tool-use +- arXiv categories: cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05363 diff --git a/papers/items/2026-2607-05378-compactionrl-reinforcement-learning-with-context-compaction-for-long-horizon-age.md b/papers/items/2026-2607-05378-compactionrl-reinforcement-learning-with-context-compaction-for-long-horizon-age.md new file mode 100644 index 0000000..46b5067 --- /dev/null +++ b/papers/items/2026-2607-05378-compactionrl-reinforcement-learning-with-context-compaction-for-long-horizon-age.md @@ -0,0 +1,60 @@ +# Paper: CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents + +--- +type: paper +title: "CompactionRL: Reinforcement Learning with Context Compaction for Long-Horizon Agents" +authors: Yujiang Li, Zhenyu Hou, Yi Jing, Jie Tang, Yuxiao Dong +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05378 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - coding-agent + - planning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: coding-agent, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent, llm-agent +- inferred topics: coding-agent, planning, tool-use +- arXiv categories: cs.LG +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05378 diff --git a/papers/items/2026-2607-05391-llm-as-a-verifier-a-general-purpose-verification-framework.md b/papers/items/2026-2607-05391-llm-as-a-verifier-a-general-purpose-verification-framework.md new file mode 100644 index 0000000..fc7fc6a --- /dev/null +++ b/papers/items/2026-2607-05391-llm-as-a-verifier-a-general-purpose-verification-framework.md @@ -0,0 +1,65 @@ +# Paper: LLM-as-a-Verifier: A General-Purpose Verification Framework + +--- +type: paper +title: "LLM-as-a-Verifier: A General-Purpose Verification Framework" +authors: Jacky Kwok, Shulu Li, Pranav Atreya, Yuejiang Liu, Yixing Jiang, Chelsea Finn, Marco Pavone, Ion Stoica, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05391 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - embodied-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - cs.LG + - cs.MA + - cs.RO +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, embodied-agent, reasoning +- arXiv categories: cs.AI, cs.CL, cs.LG, cs.MA, cs.RO +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05391 diff --git a/papers/items/2026-2607-05428-charlie-an-on-premise-multi-agent-retrieval-augmented-generation-system-for-evid.md b/papers/items/2026-2607-05428-charlie-an-on-premise-multi-agent-retrieval-augmented-generation-system-for-evid.md new file mode 100644 index 0000000..ba55ef9 --- /dev/null +++ b/papers/items/2026-2607-05428-charlie-an-on-premise-multi-agent-retrieval-augmented-generation-system-for-evid.md @@ -0,0 +1,66 @@ +# Paper: CHARLIE: An On-Premise Multi-Agent Retrieval-Augmented Generation System for Evidential Reasoning in Forensic Science + +--- +type: paper +title: "CHARLIE: An On-Premise Multi-Agent Retrieval-Augmented Generation System for Evidential Reasoning in Forensic Science" +authors: Leandro D. Carneiro, Andre L. S. Meirelles, Juliano de A. Gomes, Rafael C. A. Cabral +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05428 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-01 +updated_at: 2026-07-01 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - multi-agent + - planning + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.DL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: rag-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: rag-agent +- inferred topics: agent-evaluation, memory, multi-agent, planning, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.DL, cs.AI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05428 diff --git a/papers/items/2026-2607-05456-prompt-to-paper-agentic-ai-system-for-bioinformatics.md b/papers/items/2026-2607-05456-prompt-to-paper-agentic-ai-system-for-bioinformatics.md new file mode 100644 index 0000000..3695c17 --- /dev/null +++ b/papers/items/2026-2607-05456-prompt-to-paper-agentic-ai-system-for-bioinformatics.md @@ -0,0 +1,64 @@ +# Paper: Prompt-to-Paper: Agentic AI System for Bioinformatics + +--- +type: paper +title: "Prompt-to-Paper: Agentic AI System for Bioinformatics" +authors: Ramsha Kamran, Maheera Amjad, Zartasha Mustansar, Arsalan Shaukat, Salma Sherbaz, Muhammad U. S. Khan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05456 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL + - q-bio.QM +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: agentic-ai, coding-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, coding-agent, multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent, rag, tool-use +- arXiv categories: cs.AI, cs.CL, q-bio.QM +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05456 diff --git a/papers/items/2026-2607-05458-learning-to-control-llm-agent-harnesses-with-offline-reinforcement-learning.md b/papers/items/2026-2607-05458-learning-to-control-llm-agent-harnesses-with-offline-reinforcement-learning.md new file mode 100644 index 0000000..6e66037 --- /dev/null +++ b/papers/items/2026-2607-05458-learning-to-control-llm-agent-harnesses-with-offline-reinforcement-learning.md @@ -0,0 +1,63 @@ +# Paper: Learning to Control LLM Agent Harnesses with Offline Reinforcement Learning + +--- +type: paper +title: Learning to Control LLM Agent Harnesses with Offline Reinforcement Learning +authors: Haiwen Yi, Xinyuan Song +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05458 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-05 +updated_at: 2026-07-05 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.LG + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, coding-agent, reasoning, tool-use, workflow-agent +- arXiv categories: cs.LG, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05458 diff --git a/papers/items/2026-2607-05518-aiauthz-off-host-identity-bound-authorization-for-ai-agents.md b/papers/items/2026-2607-05518-aiauthz-off-host-identity-bound-authorization-for-ai-agents.md new file mode 100644 index 0000000..255c81e --- /dev/null +++ b/papers/items/2026-2607-05518-aiauthz-off-host-identity-bound-authorization-for-ai-agents.md @@ -0,0 +1,61 @@ +# Paper: aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents + +--- +type: paper +title: "aiAuthZ: Off-Host, Identity-Bound Authorization for AI Agents" +authors: Sai Varun Kodathala +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05518 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent +- inferred topics: agent-evaluation, agent-safety, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05518 diff --git a/papers/items/2026-2607-05659-agents-with-feelings-personality-and-emotion-in-multi-agent-software-teams.md b/papers/items/2026-2607-05659-agents-with-feelings-personality-and-emotion-in-multi-agent-software-teams.md new file mode 100644 index 0000000..172ca12 --- /dev/null +++ b/papers/items/2026-2607-05659-agents-with-feelings-personality-and-emotion-in-multi-agent-software-teams.md @@ -0,0 +1,61 @@ +# Paper: Agents with Feelings? Personality and Emotion in Multi-Agent Software Teams + +--- +type: paper +title: Agents with Feelings? Personality and Emotion in Multi-Agent Software Teams +authors: Yunyan Ding, Thomas Zimmermann, Iftekhar Ahmed +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05659 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - multi-agent + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, coding-agent, multi-agent, workflow-agent +- arXiv categories: cs.SE +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05659 diff --git a/papers/items/2026-2607-05666-what-do-ai-agents-actually-change-an-empirical-taxonomy-of-mutation-patterns-in-.md b/papers/items/2026-2607-05666-what-do-ai-agents-actually-change-an-empirical-taxonomy-of-mutation-patterns-in-.md new file mode 100644 index 0000000..fa057ed --- /dev/null +++ b/papers/items/2026-2607-05666-what-do-ai-agents-actually-change-an-empirical-taxonomy-of-mutation-patterns-in-.md @@ -0,0 +1,59 @@ +# Paper: What Do AI Agents Actually Change? An Empirical Taxonomy of Mutation Patterns in Performance-Improving Pull Requests + +--- +type: paper +title: What Do AI Agents Actually Change? An Empirical Taxonomy of Mutation Patterns in Performance-Improving Pull Requests +authors: Illia Dovhoshliubnyi, Nima Soroush, Ashkan Sami, Alexander Brownlee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05666 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - coding-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: ai-agent, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: ai-agent, coding-agent +- inferred topics: coding-agent +- arXiv categories: cs.SE, cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05666 diff --git a/papers/items/2026-2607-05677-from-conversation-to-contribution-characterizing-coding-agent-in-open-source-sof.md b/papers/items/2026-2607-05677-from-conversation-to-contribution-characterizing-coding-agent-in-open-source-sof.md new file mode 100644 index 0000000..163e50b --- /dev/null +++ b/papers/items/2026-2607-05677-from-conversation-to-contribution-characterizing-coding-agent-in-open-source-sof.md @@ -0,0 +1,64 @@ +# Paper: From Conversation to Contribution: Characterizing Coding Agent in Open-Source Software + +--- +type: paper +title: "From Conversation to Contribution: Characterizing Coding Agent in Open-Source Software" +authors: Zihan Fang, Yueke Zhang, Ningzhi Tang, Collin McMillan, Toby Jia-Jun Li, Yu Huang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05677 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - computer-use + - multi-agent + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, computer-use, multi-agent, tool-use, workflow-agent +- arXiv categories: cs.SE, cs.HC +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05677 diff --git a/papers/items/2026-2607-05690-memory-in-the-loop-in-process-retrieval-as-extendedworking-memory-for-language-a.md b/papers/items/2026-2607-05690-memory-in-the-loop-in-process-retrieval-as-extendedworking-memory-for-language-a.md new file mode 100644 index 0000000..0fe57af --- /dev/null +++ b/papers/items/2026-2607-05690-memory-in-the-loop-in-process-retrieval-as-extendedworking-memory-for-language-a.md @@ -0,0 +1,62 @@ +# Paper: Memory in the Loop: In-Process Retrieval as ExtendedWorking Memory for Language Agents + +--- +type: paper +title: "Memory in the Loop: In-Process Retrieval as ExtendedWorking Memory for Language Agents" +authors: Yusuf Khan, Carlo Lipizzi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05690 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-06 +updated_at: 2026-07-06 +status: queued +relevance: high +topics: + - agent-evaluation + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: language-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: language-agent +- inferred topics: agent-evaluation, memory, rag, tool-use +- arXiv categories: cs.AI, cs.CL +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05690 diff --git a/papers/items/2026-2607-05743-the-balkanization-of-execution-security-research-for-ai-coding-agents-isolation-.md b/papers/items/2026-2607-05743-the-balkanization-of-execution-security-research-for-ai-coding-agents-isolation-.md new file mode 100644 index 0000000..4f59bc7 --- /dev/null +++ b/papers/items/2026-2607-05743-the-balkanization-of-execution-security-research-for-ai-coding-agents-isolation-.md @@ -0,0 +1,62 @@ +# Paper: The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities + +--- +type: paper +title: "The Balkanization of Execution-Security Research for AI Coding Agents: Isolation, Access Control, and Time-of-Check-to-Time-of-Use Vulnerabilities" +authors: Mohammadreza Rashidi +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05743 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, coding-agent +- inferred topics: agent-evaluation, agent-safety, coding-agent, tool-use +- arXiv categories: cs.CR, cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05743 diff --git a/papers/items/2026-2607-05772-detecting-vulnerability-inducing-commits-via-multi-stage-reasoning-with-llm-base.md b/papers/items/2026-2607-05772-detecting-vulnerability-inducing-commits-via-multi-stage-reasoning-with-llm-base.md new file mode 100644 index 0000000..9896f02 --- /dev/null +++ b/papers/items/2026-2607-05772-detecting-vulnerability-inducing-commits-via-multi-stage-reasoning-with-llm-base.md @@ -0,0 +1,63 @@ +# Paper: Detecting Vulnerability-Inducing Commits via Multi-Stage Reasoning with LLM-Based Agents + +--- +type: paper +title: Detecting Vulnerability-Inducing Commits via Multi-Stage Reasoning with LLM-Based Agents +authors: Liyou Chen, Hailong Sun, Xiang Gao, Yue Pan +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05772 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-safety + - multi-agent + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-safety, multi-agent, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.SE +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05772 diff --git a/papers/items/2026-2607-05773-beyond-static-evaluation-building-simulation-environments-for-scalable-agentic-r.md b/papers/items/2026-2607-05773-beyond-static-evaluation-building-simulation-environments-for-scalable-agentic-r.md new file mode 100644 index 0000000..0ffa378 --- /dev/null +++ b/papers/items/2026-2607-05773-beyond-static-evaluation-building-simulation-environments-for-scalable-agentic-r.md @@ -0,0 +1,61 @@ +# Paper: Beyond Static Evaluation: Building Simulation Environments for Scalable Agentic Reinforcement Learning + +--- +type: paper +title: "Beyond Static Evaluation: Building Simulation Environments for Scalable Agentic Reinforcement Learning" +authors: Akshay Arora, Ishan Nigam, Ashutosh Aggarwal, Shefali Bansal, Krishna Singh, Sweta Kumari, Nikhil Mittal, Shariq Farhan, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05773 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: autonomous-agent-llm, tool-use, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: autonomous-agent-llm, tool-use, web-gui-agent +- inferred topics: agent-evaluation, computer-use, tool-use, world-model +- arXiv categories: cs.AI +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05773 diff --git a/papers/items/2026-2607-05775-beyond-the-leaderboard-a-synthesis-of-tool-use-planning-and-reasoning-failures-i.md b/papers/items/2026-2607-05775-beyond-the-leaderboard-a-synthesis-of-tool-use-planning-and-reasoning-failures-i.md new file mode 100644 index 0000000..930596d --- /dev/null +++ b/papers/items/2026-2607-05775-beyond-the-leaderboard-a-synthesis-of-tool-use-planning-and-reasoning-failures-i.md @@ -0,0 +1,65 @@ +# Paper: Beyond the Leaderboard: A Synthesis of Tool-Use, Planning, and Reasoning Failures in Large Language Model Agents + +--- +type: paper +title: "Beyond the Leaderboard: A Synthesis of Tool-Use, Planning, and Reasoning Failures in Large Language Model Agents" +authors: Wael Albayaydh, Rui Zhao, Ivan Flechais +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05775 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - coding-agent + - embodied-agent + - multi-agent + - planning + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 23 +collection_queries: llm-agent, multi-agent-llm, planning-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm, planning-agent, tool-use +- inferred topics: agent-evaluation, agent-safety, coding-agent, embodied-agent, multi-agent, planning, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 23 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05775 diff --git a/papers/items/2026-2607-05794-from-passive-retrieval-to-active-memory-navigation-learning-to-use-memory-as-a-s.md b/papers/items/2026-2607-05794-from-passive-retrieval-to-active-memory-navigation-learning-to-use-memory-as-a-s.md new file mode 100644 index 0000000..625166d --- /dev/null +++ b/papers/items/2026-2607-05794-from-passive-retrieval-to-active-memory-navigation-learning-to-use-memory-as-a-s.md @@ -0,0 +1,63 @@ +# Paper: From Passive Retrieval to Active Memory Navigation: Learning to Use Memory as a Structured Action Space + +--- +type: paper +title: "From Passive Retrieval to Active Memory Navigation: Learning to Use Memory as a Structured Action Space" +authors: Yue Xu, Yutao Sun, Yihao Liu, Mengyu Zhou, Jiayi Qiao, Lu Ma, Kai Tang, Wenjie Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05794 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - embodied-agent + - memory + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: tool-use +- inferred topics: agent-evaluation, embodied-agent, memory, rag, reasoning, tool-use +- arXiv categories: cs.AI +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05794 diff --git a/papers/items/2026-2607-05805-onnes-a-physics-grounded-multi-agent-llm-simulator-for-cryogenic-fault-diagnosis.md b/papers/items/2026-2607-05805-onnes-a-physics-grounded-multi-agent-llm-simulator-for-cryogenic-fault-diagnosis.md new file mode 100644 index 0000000..a33917f --- /dev/null +++ b/papers/items/2026-2607-05805-onnes-a-physics-grounded-multi-agent-llm-simulator-for-cryogenic-fault-diagnosis.md @@ -0,0 +1,61 @@ +# Paper: Onnes: A Physics-Grounded Multi-Agent LLM Simulator for Cryogenic Fault Diagnosis in Quantum Computing Infrastructure + +--- +type: paper +title: "Onnes: A Physics-Grounded Multi-Agent LLM Simulator for Cryogenic Fault Diagnosis in Quantum Computing Infrastructure" +authors: Praneeth Narisetty, Uday Kumar Reddy Kattamanchi, Shiva Nagendra Babu Kore +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05805 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.LG + - quant-ph +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, multi-agent +- arXiv categories: cs.AI, cs.LG, quant-ph +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05805 diff --git a/papers/items/2026-2607-05915-pcbworld-a-benchmark-environment-for-engine-grounded-pcb-design-automation.md b/papers/items/2026-2607-05915-pcbworld-a-benchmark-environment-for-engine-grounded-pcb-design-automation.md new file mode 100644 index 0000000..678969c --- /dev/null +++ b/papers/items/2026-2607-05915-pcbworld-a-benchmark-environment-for-engine-grounded-pcb-design-automation.md @@ -0,0 +1,60 @@ +# Paper: PCBWorld: A Benchmark Environment for Engine-Grounded PCB Design Automation + +--- +type: paper +title: "PCBWorld: A Benchmark Environment for Engine-Grounded PCB Design Automation" +authors: Hyungseok Song, Junseok Park, Won-Seok Choi, Seohui Bae, Han-Seul Jeong, Youngjoon Park, Soonyoung Lee +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.05915 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: agentic-ai, llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, llm-agent, tool-use +- inferred topics: agent-evaluation, tool-use, workflow-agent +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.05915 diff --git a/papers/items/2026-2607-06000-context-to-execution-integrity-for-llm-agents.md b/papers/items/2026-2607-06000-context-to-execution-integrity-for-llm-agents.md new file mode 100644 index 0000000..9ce7fb2 --- /dev/null +++ b/papers/items/2026-2607-06000-context-to-execution-integrity-for-llm-agents.md @@ -0,0 +1,60 @@ +# Paper: Context-to-Execution Integrity for LLM Agents + +--- +type: paper +title: Context-to-Execution Integrity for LLM Agents +authors: Igor Santos-Grueiro +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06000 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CR +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation, coding-agent, llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, coding-agent, llm-agent +- inferred topics: agent-evaluation, coding-agent, tool-use +- arXiv categories: cs.CR +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06000 diff --git a/papers/items/2026-2607-06001-information-limits-and-attractor-dynamics-in-economies-of-frontier-llm-agents-a-.md b/papers/items/2026-2607-06001-information-limits-and-attractor-dynamics-in-economies-of-frontier-llm-agents-a-.md new file mode 100644 index 0000000..e609671 --- /dev/null +++ b/papers/items/2026-2607-06001-information-limits-and-attractor-dynamics-in-economies-of-frontier-llm-agents-a-.md @@ -0,0 +1,62 @@ +# Paper: Information Limits and Attractor Dynamics in Economies of Frontier LLM Agents: A Pre-Registered Test + +--- +type: paper +title: "Information Limits and Attractor Dynamics in Economies of Frontier LLM Agents: A Pre-Registered Test" +authors: Cheng Qian +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06001 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-safety + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.MA +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-safety, multi-agent, reasoning, tool-use +- arXiv categories: cs.AI, cs.MA +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06001 diff --git a/papers/items/2026-2607-06008-polyworkbench-benchmarking-multilingual-long-horizon-llm-agents.md b/papers/items/2026-2607-06008-polyworkbench-benchmarking-multilingual-long-horizon-llm-agents.md new file mode 100644 index 0000000..be00e09 --- /dev/null +++ b/papers/items/2026-2607-06008-polyworkbench-benchmarking-multilingual-long-horizon-llm-agents.md @@ -0,0 +1,64 @@ +# Paper: PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents + +--- +type: paper +title: "PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents" +authors: Hongliang Li, Yijin Liu, Zhiwei Zhang, Zihe Liu, Xinyue Lou, Jinan Xu, Fandong Meng, Kaiyu Huang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06008 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 25 +collection_queries: agent-evaluation, llm-agent, planning-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, llm-agent, planning-agent, tool-use +- inferred topics: agent-evaluation, computer-use, planning, reasoning, tool-use, workflow-agent +- arXiv categories: cs.AI, cs.CL +- collection score: 25 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06008 diff --git a/papers/items/2026-2607-06080-from-blueprint-to-reality-modeling-and-applying-putnam-s-social-capital-theory-w.md b/papers/items/2026-2607-06080-from-blueprint-to-reality-modeling-and-applying-putnam-s-social-capital-theory-w.md new file mode 100644 index 0000000..54488db --- /dev/null +++ b/papers/items/2026-2607-06080-from-blueprint-to-reality-modeling-and-applying-putnam-s-social-capital-theory-w.md @@ -0,0 +1,64 @@ +# Paper: From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations + +--- +type: paper +title: "From Blueprint to Reality: Modeling and Applying Putnam's Social Capital Theory with LLM-based Multi-agent Simulations" +authors: Shiyi Ling, Zhi Zheng, Hui Zheng, Wenjun Xue, Feng Ye, Tong Xu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06080 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-safety + - multi-agent + - rag + - tool-use + - world-model +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI + - cs.SI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-safety, multi-agent, rag, tool-use, world-model +- arXiv categories: cs.CL, cs.AI, cs.SI +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06080 diff --git a/papers/items/2026-2607-06101-agents-that-teach-towards-designing-incidental-learning-back-into-ai-assisted-so.md b/papers/items/2026-2607-06101-agents-that-teach-towards-designing-incidental-learning-back-into-ai-assisted-so.md new file mode 100644 index 0000000..86fdc04 --- /dev/null +++ b/papers/items/2026-2607-06101-agents-that-teach-towards-designing-incidental-learning-back-into-ai-assisted-so.md @@ -0,0 +1,68 @@ +# Paper: Agents That Teach: Towards Designing Incidental Learning Back into AI-Assisted Software Development + +--- +type: paper +title: "Agents That Teach: Towards Designing Incidental Learning Back into AI-Assisted Software Development" +authors: Rohit Mehra, Samdyuti Suri, Prithviraj K Tagadinamani, Kapil Singi, Vikrant Kaulgud, Adam P. Burden +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06101 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-safety + - coding-agent + - computer-use + - multi-agent + - rag + - reasoning + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CY + - cs.HC +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-safety, coding-agent, computer-use, multi-agent, rag, reasoning, tool-use, workflow-agent +- arXiv categories: cs.SE, cs.AI, cs.CY, cs.HC +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06101 diff --git a/papers/items/2026-2607-06118-webretriever-a-large-scale-comprehensive-benchmark-for-efficient-web-agent-evalu.md b/papers/items/2026-2607-06118-webretriever-a-large-scale-comprehensive-benchmark-for-efficient-web-agent-evalu.md new file mode 100644 index 0000000..c89f63f --- /dev/null +++ b/papers/items/2026-2607-06118-webretriever-a-large-scale-comprehensive-benchmark-for-efficient-web-agent-evalu.md @@ -0,0 +1,65 @@ +# Paper: WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation + +--- +type: paper +title: "WebRetriever: A Large-Scale Comprehensive Benchmark for Efficient Web Agent Evaluation" +authors: Wei Dong, Tianyu Fu, Zhe Yu, Hanning Wang, Anyang Su, Zhizhou Fang, Yuyang Chen, Shuo Wang, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06118 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - agent-safety + - computer-use + - embodied-agent + - rag + - tool-use + - workflow-agent +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CV + - cs.MM +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 21 +collection_queries: agent-evaluation, web-gui-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, web-gui-agent +- inferred topics: agent-evaluation, agent-safety, computer-use, embodied-agent, rag, tool-use, workflow-agent +- arXiv categories: cs.CV, cs.MM +- collection score: 21 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06118 diff --git a/papers/items/2026-2607-06140-curateevo-data-curation-evolving-for-agentic-post-training.md b/papers/items/2026-2607-06140-curateevo-data-curation-evolving-for-agentic-post-training.md new file mode 100644 index 0000000..d6094d5 --- /dev/null +++ b/papers/items/2026-2607-06140-curateevo-data-curation-evolving-for-agentic-post-training.md @@ -0,0 +1,60 @@ +# Paper: CurateEvo: Data-Curation Evolving for Agentic Post-Training + +--- +type: paper +title: "CurateEvo: Data-Curation Evolving for Agentic Post-Training" +authors: Dingzirui Wang, Xuanliang Zhang, Keyan Xu, Qingfu Zhu, Wanxiang Che +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06140 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - memory + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: memory, planning, rag +- arXiv categories: cs.CL +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06140 diff --git a/papers/items/2026-2607-06157-llm-agents-for-deliberative-collaboration-a-study-on-joint-decision-making-under.md b/papers/items/2026-2607-06157-llm-agents-for-deliberative-collaboration-a-study-on-joint-decision-making-under.md new file mode 100644 index 0000000..b8bf994 --- /dev/null +++ b/papers/items/2026-2607-06157-llm-agents-for-deliberative-collaboration-a-study-on-joint-decision-making-under.md @@ -0,0 +1,62 @@ +# Paper: LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability + +--- +type: paper +title: "LLM Agents for Deliberative Collaboration: A Study on Joint Decision Making Under Partial Observability" +authors: Chenxu Wang, Yongkun Yang, Boyuan Du, Shiwei Lin, Huaping Liu +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06157 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 18 +collection_queries: llm-agent, multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, reasoning, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 18 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06157 diff --git a/papers/items/2026-2607-06195-logichunter-testing-llm-agent-frameworks-with-an-agentic-oracle.md b/papers/items/2026-2607-06195-logichunter-testing-llm-agent-frameworks-with-an-agentic-oracle.md new file mode 100644 index 0000000..986e544 --- /dev/null +++ b/papers/items/2026-2607-06195-logichunter-testing-llm-agent-frameworks-with-an-agentic-oracle.md @@ -0,0 +1,60 @@ +# Paper: LogicHunter: Testing LLM Agent Frameworks with an Agentic Oracle + +--- +type: paper +title: "LogicHunter: Testing LLM Agent Frameworks with an Agentic Oracle" +authors: Minghui Long, Yanjie Zhao, Haoyu Wang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06195 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, computer-use, memory +- arXiv categories: cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06195 diff --git a/papers/items/2026-2607-06223-information-gain-based-rollout-policy-optimization-an-adaptive-tree-structured-r.md b/papers/items/2026-2607-06223-information-gain-based-rollout-policy-optimization-an-adaptive-tree-structured-r.md new file mode 100644 index 0000000..6515f0b --- /dev/null +++ b/papers/items/2026-2607-06223-information-gain-based-rollout-policy-optimization-an-adaptive-tree-structured-r.md @@ -0,0 +1,61 @@ +# Paper: Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents + +--- +type: paper +title: "Information Gain-based Rollout Policy Optimization: An Adaptive Tree-Structured Rollout Approach for Multi-Turn LLM Agents" +authors: Yijun Zhang, Fan Xu, Jiaxin Ding, Yule Xie, Shiqing Gao, Xin Ding, Haoxiang Zhang, Luoyi Fu, et al. +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06223 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - planning + - rag +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 15 +collection_queries: llm-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent +- inferred topics: agent-evaluation, computer-use, planning, rag +- arXiv categories: cs.AI +- collection score: 15 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06223 diff --git a/papers/items/2026-2607-06273-agenttether-graph-guided-diagnosis-and-runtime-intervention-for-reliable-llm-age.md b/papers/items/2026-2607-06273-agenttether-graph-guided-diagnosis-and-runtime-intervention-for-reliable-llm-age.md new file mode 100644 index 0000000..b2cfee5 --- /dev/null +++ b/papers/items/2026-2607-06273-agenttether-graph-guided-diagnosis-and-runtime-intervention-for-reliable-llm-age.md @@ -0,0 +1,62 @@ +# Paper: AgentTether: Graph-Guided Diagnosis and Runtime Intervention for Reliable LLM Agent Operation + +--- +type: paper +title: "AgentTether: Graph-Guided Diagnosis and Runtime Intervention for Reliable LLM Agent Operation" +authors: Chenyu Zhao, Shenglin Zhang, Wenwei Gu, Yongqian Sun, Dan Pei, Chetan Bansal, Saravan Rajmohan, Minghua Ma +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06273 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - computer-use + - memory + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 17 +collection_queries: llm-agent, tool-use +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: llm-agent, tool-use +- inferred topics: agent-evaluation, computer-use, memory, reasoning, tool-use +- arXiv categories: cs.SE +- collection score: 17 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06273 diff --git a/papers/items/2026-2607-06341-harnessing-code-agents-for-automatic-software-verification.md b/papers/items/2026-2607-06341-harnessing-code-agents-for-automatic-software-verification.md new file mode 100644 index 0000000..3ed1aff --- /dev/null +++ b/papers/items/2026-2607-06341-harnessing-code-agents-for-automatic-software-verification.md @@ -0,0 +1,64 @@ +# Paper: Harnessing Code Agents for Automatic Software Verification + +--- +type: paper +title: Harnessing Code Agents for Automatic Software Verification +authors: Shuangxiang Kan, Shuanglong Kan, Sebastian Ertel +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06341 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - memory + - rag + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.FL + - cs.AI + - cs.SE +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 13 +collection_queries: coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: coding-agent +- inferred topics: agent-evaluation, coding-agent, memory, rag, tool-use +- arXiv categories: cs.FL, cs.AI, cs.SE +- collection score: 13 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06341 diff --git a/papers/items/2026-2607-06411-rubench-a-repository-level-agentic-coding-benchmark-with-natively-authored-russi.md b/papers/items/2026-2607-06411-rubench-a-repository-level-agentic-coding-benchmark-with-natively-authored-russi.md new file mode 100644 index 0000000..250cd26 --- /dev/null +++ b/papers/items/2026-2607-06411-rubench-a-repository-level-agentic-coding-benchmark-with-natively-authored-russi.md @@ -0,0 +1,62 @@ +# Paper: RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications + +--- +type: paper +title: "RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications" +authors: Evgeny Shilov +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06411 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.SE + - cs.AI + - cs.CL +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agent-evaluation, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agent-evaluation, coding-agent +- inferred topics: agent-evaluation, coding-agent, reasoning +- arXiv categories: cs.SE, cs.AI, cs.CL +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06411 diff --git a/papers/items/2026-2607-06413-an-experimental-design-approach-to-evaluating-agentic-ai-s-autonomous-model-disc.md b/papers/items/2026-2607-06413-an-experimental-design-approach-to-evaluating-agentic-ai-s-autonomous-model-disc.md new file mode 100644 index 0000000..aede1e9 --- /dev/null +++ b/papers/items/2026-2607-06413-an-experimental-design-approach-to-evaluating-agentic-ai-s-autonomous-model-disc.md @@ -0,0 +1,61 @@ +# Paper: An Experimental Design Approach to Evaluating Agentic AI's Autonomous Model Discovery + +--- +type: paper +title: "An Experimental Design Approach to Evaluating Agentic AI's Autonomous Model Discovery" +authors: Hao He, Xueying Liu, Chris J. Kuhlman, Xinwei Deng +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06413 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - coding-agent + - reasoning +methods: + - +benchmarks: + - +models: + - +datasets: + - stat.ME + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 16 +collection_queries: agentic-ai, coding-agent +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: agentic-ai, coding-agent +- inferred topics: agent-evaluation, coding-agent, reasoning +- arXiv categories: stat.ME, cs.AI +- collection score: 16 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06413 diff --git a/papers/items/2026-2607-06452-from-voting-to-agent-collaboration-answer-type-aware-llm-pipelines-for-bioasq-14.md b/papers/items/2026-2607-06452-from-voting-to-agent-collaboration-answer-type-aware-llm-pipelines-for-bioasq-14.md new file mode 100644 index 0000000..768a8a8 --- /dev/null +++ b/papers/items/2026-2607-06452-from-voting-to-agent-collaboration-answer-type-aware-llm-pipelines-for-bioasq-14.md @@ -0,0 +1,63 @@ +# Paper: From Voting to Agent Collaboration: Answer-Type-Aware LLM Pipelines for BioASQ 14b + +--- +type: paper +title: "From Voting to Agent Collaboration: Answer-Type-Aware LLM Pipelines for BioASQ 14b" +authors: Taeyun Roh, Eunha Lee, Wonjune Jang, Sohyun Chung, Junha Jung, Jaewoo Kang +year: 2026 +venue: arXiv +url: https://arxiv.org/abs/2607.06452 +code_url: +source: arxiv +collected_at: 2026-07-08 +published_at: 2026-07-07 +updated_at: 2026-07-07 +status: queued +relevance: high +topics: + - agent-evaluation + - multi-agent + - rag + - reasoning + - tool-use +methods: + - +benchmarks: + - +models: + - +datasets: + - cs.CL + - cs.AI +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: 14 +collection_queries: multi-agent-llm +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: multi-agent-llm +- inferred topics: agent-evaluation, multi-agent, rag, reasoning, tool-use +- arXiv categories: cs.CL, cs.AI +- collection score: 14 + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: https://arxiv.org/abs/2607.06452 diff --git a/papers/paper-insights.md b/papers/paper-insights.md index d6e0ae9..3ad63e2 100644 --- a/papers/paper-insights.md +++ b/papers/paper-insights.md @@ -7,9 +7,13 @@ status: seed ## Current Snapshot - date: 2026-07-08 -- scope: Agent evaluation, memory, coding-agent benchmarks -- papers reviewed: 6 starter items -- caveat: sources are provisional; source approval workflow still needs user confirmation. +- scope: arXiv Agent-related papers from 2025-07-08 to 2026-07-08 +- candidates seen: 1506 +- high-relevance candidates selected into paper items: 970 +- reserve candidates retained in manifest: 536 +- total paper items: 974 +- detailed summary: [Paper Corpus Summary](corpus-summary-2026-07-08.md) +- caveat: high-recall automated sweep; most items are `queued` and need human skim. ## Strong Signals @@ -17,12 +21,15 @@ status: seed - Memory 不能只看事实召回;更重要的是是否改善流程执行和跨任务决策。 - Coding-agent benchmark 正在从单 issue patch 转向长程软件演进和人机对话。 - JD 高频词和论文趋势高度重合:memory、eval、planning/tool use、coding agent。 +- 最近一年 Agent 论文的重心明显落在 evaluation、tool use、memory、safety、coding agent、computer-use/GUI agent。 +- 2026-05 到 2026-07 的论文密度明显上升,说明 Agent research 正在快速扩张。 ## Weak or Unproven Claims - “加 memory 就更强”不成立;EvoMemBench 显示长上下文 baseline 很强,memory 价值依赖任务。 - 单一 safety benchmark 不足以说明 Agent 安全;不同 benchmark 的方法差异可能改变结论。 - SWE-Bench 类单点任务不足以代表真实 coding agent 生产力。 +- 自动标签是召回导向,不能直接当作精确分类;后续需要人工把高价值论文升到 `skimmed`。 ## Mature Directions @@ -30,6 +37,8 @@ status: seed - Memory benchmark and memory architecture comparison - Coding agent benchmark beyond SWE-Bench - Agent failure diagnosis and trajectory analysis +- Runtime governance and agent security +- Computer-use / GUI agent benchmark ## Emerging Directions @@ -37,6 +46,10 @@ status: seed - Stateful enterprise task benchmarks - Agent safety benchmark taxonomy - Skill/prompt optimization as trainable artifacts +- Agent memory poisoning and memory provenance +- Workflow-agent and enterprise automation benchmarks +- Multi-agent shared memory and governance +- Agentic RAG for technical/scientific literature ## Experiment Candidates @@ -46,10 +59,12 @@ status: seed | Memory vs Long Context | Memory survey, EvoMemBench, STATE-Bench | success rate, token cost, latency, contradiction rate | 必须包含 long-context baseline | | Long-horizon Coding Eval | SWE-EVO, Dialogue-SWEBench | task success, fix rate, clarification quality | 从真实 repo release notes 构造小任务 | | Failure Step Diagnosis | AgentRx, OpenAgentSafety | critical failure step accuracy | 需要统一 trace schema | +| GUI Agent Safety | OSWorld2.0, MacAgentBench, computer-use papers | irreversible action rate, prompt-injection catch rate | 需要 sandbox 和人工确认策略 | +| Agent Memory Security | memory poisoning / forged reasoning papers | poisoned recall rate, provenance coverage | 和长期记忆系统强相关 | ## Links Back to Knowledge Base - docs to update: evaluation, memory, coding-agent, safety -- experiments to create: memory comparison, mini safety suite, coding-agent eval +- experiments to create: memory comparison, mini safety suite, coding-agent eval, GUI-agent safety eval - projects affected: future front-end/indexing project, coding-agent experiments - job skills connected: agent-evaluation, memory, coding-agent, tool-use, planning diff --git a/papers/source-registry.md b/papers/source-registry.md index 3c776b7..9b2ec04 100644 --- a/papers/source-registry.md +++ b/papers/source-registry.md @@ -8,7 +8,7 @@ | Source | Status | Notes | | --- | --- | --- | -| arXiv | provisional-used | 新论文发现和预印本;first sweep 已使用,待用户正式确认 | +| arXiv | provisional-used | 新论文发现和预印本;2026-07-08 扩召回看到 1506 个候选,970 个 high-relevance 入库,536 个留作 reserve | | OpenReview | candidate | ICLR、NeurIPS 等投稿和评审线索 | | ACL Anthology | candidate | NLP、RAG、tool use、evaluation 相关论文 | | Conference proceedings | candidate | NeurIPS、ICLR、ICML、ACL、EMNLP、KDD、WWW 等 | diff --git a/tools/collection/README.md b/tools/collection/README.md index a7cc60f..b5efdcb 100644 --- a/tools/collection/README.md +++ b/tools/collection/README.md @@ -43,6 +43,26 @@ python3 tools/collection/check_urls.py 可加 `--json` 输出机器可读结果。 +### `collect_arxiv.py` + +用 arXiv API 批量收集最近一年 Agent 相关论文,生成 `papers/items/` 条目、`data/arxiv-agent-papers-*.json` 入库清单和 `data/arxiv-agent-candidates-*.json` 全量候选清单。 + +```bash +python3 tools/collection/collect_arxiv.py \ + --from-date 2025-07-08 \ + --to-date 2026-07-08 \ + --max-items 180 \ + --sleep 1.0 +``` + +默认策略: + +- 多组关键词召回:LLM agent、language agent、tool use、memory、evaluation、coding agent、web/GUI agent、multi-agent、safety、RAG 等。 +- 自动去重:按 arXiv ID 去重。 +- 自动打标签:根据标题和摘要推断 topics。 +- 自动分层:生成 `queued` 条目,后续人工精读后再升级状态。 +- API 退避:遇到 arXiv 429 或临时网络错误时按 `--retries` 和 `--retry-sleep` 重试。 + ## Method 1. 搜索或导入资料。 @@ -51,3 +71,22 @@ python3 tools/collection/check_urls.py 4. 运行 `build_index.py` 更新结构化数据。 5. 运行 `check_urls.py` 检查来源可达性。 6. 更新对应 insight 文件。 + +大规模论文扩展时,先运行 `collect_arxiv.py`,再运行 `build_index.py`。 + +### `promote_arxiv_manifest.py` + +从 `collect_arxiv.py` 生成的候选池里,把达到阈值的记录批量提升为 `papers/items/` 条目。适合在 arXiv API 限流后继续本地调阈值。 + +```bash +python3 tools/collection/promote_arxiv_manifest.py \ + data/arxiv-agent-candidates-2025-07-08-to-2026-07-08.json \ + --min-score 13 \ + --relevance high \ + --output-manifest data/arxiv-agent-papers-2025-07-08-to-2026-07-08.json +``` + +默认用途: + +- `score >= 13` 且 `relevance=high` 的论文进入主阅读队列。 +- 低分候选仍留在 candidate manifest,后续可以人工抽查或降低阈值再提升。 diff --git a/tools/collection/collect_arxiv.py b/tools/collection/collect_arxiv.py new file mode 100644 index 0000000..9f41649 --- /dev/null +++ b/tools/collection/collect_arxiv.py @@ -0,0 +1,425 @@ +#!/usr/bin/env python3 +"""Collect recent arXiv papers related to LLM/AI agents.""" + +from __future__ import annotations + +import argparse +import json +import re +import time +import urllib.parse +import urllib.request +import urllib.error +import xml.etree.ElementTree as ET +from collections import defaultdict +from datetime import date +from pathlib import Path + + +ROOT = Path(__file__).resolve().parents[2] +API_URL = "https://export.arxiv.org/api/query" +NS = { + "atom": "http://www.w3.org/2005/Atom", + "arxiv": "http://arxiv.org/schemas/atom", +} + + +QUERIES = [ + ("llm-agent", 'all:"LLM agent" OR all:"LLM agents"'), + ("language-agent", 'all:"language agent" OR all:"language agents"'), + ("ai-agent", 'all:"AI agent" OR all:"AI agents"'), + ("agentic-ai", 'all:"agentic AI" OR all:"agentic workflow"'), + ("agent-evaluation", 'all:"agent evaluation" OR all:"agent benchmark" OR all:"agentic benchmark"'), + ("agent-memory", 'all:"agent memory" OR all:"memory agent" OR all:"memory agents"'), + ("tool-use", 'all:"tool use" AND all:"agent"'), + ("function-calling", 'all:"function calling" AND all:"agent"'), + ("coding-agent", 'all:"coding agent" OR all:"software engineering agent" OR all:"SWE-bench"'), + ("web-gui-agent", 'all:"web agent" OR all:"browser agent" OR all:"GUI agent" OR all:"computer use"'), + ("multi-agent-llm", 'all:"multi-agent" AND (all:"LLM" OR all:"large language model")'), + ("agent-safety", 'all:"agent safety" OR all:"AI agent safety" OR all:"agent security"'), + ("rag-agent", 'all:"RAG" AND all:"agent"'), + ("planning-agent", 'all:"planning" AND all:"LLM agent"'), + ("autonomous-agent-llm", 'all:"autonomous agent" AND (all:"LLM" OR all:"large language model")'), +] + + +TOPIC_RULES = [ + ("agent-evaluation", ("evaluation", "benchmark", "eval", "metric", "leaderboard", "assessment")), + ("memory", ("memory", "remember", "stateful", "long-term", "episodic")), + ("tool-use", ("tool", "function calling", "api", "mcp", "action")), + ("coding-agent", ("coding", "software engineering", "swe-bench", "repository", "program repair", "code agent")), + ("computer-use", ("computer use", "gui", "browser", "web agent", "mobile", "desktop")), + ("multi-agent", ("multi-agent", "multiagent", "agent society", "collaboration", "debate")), + ("agent-safety", ("safety", "security", "risk", "prompt injection", "alignment", "guardrail")), + ("rag", ("retrieval", "rag", "knowledge base", "grounding")), + ("planning", ("planning", "planner", "plan", "long-horizon", "task decomposition")), + ("reasoning", ("reasoning", "reflection", "self-improvement", "verifier")), + ("workflow-agent", ("workflow", "enterprise", "office", "productivity", "automation")), + ("embodied-agent", ("embodied", "robot", "robotic", "navigation")), + ("world-model", ("world model", "simulation", "environment model")), +] + + +HIGH_SIGNAL_TERMS = ( + "llm agent", + "language agent", + "ai agent", + "agentic", + "tool use", + "memory", + "benchmark", + "evaluation", + "coding agent", + "swe-bench", + "computer use", + "gui agent", + "multi-agent", + "agent safety", +) + + +def normalize_space(value: str) -> str: + return re.sub(r"\s+", " ", value).strip() + + +def slugify(text: str) -> str: + text = text.lower() + text = re.sub(r"[^a-z0-9]+", "-", text) + return re.sub(r"-+", "-", text).strip("-")[:80] or "untitled" + + +def yaml_value(value: str | int | None) -> str: + if value is None: + return "" + text = str(value).replace("\n", " ").strip() + if not text: + return "" + if any(char in text for char in [":", "#", "[", "]", "{", "}", "\"", "'"]): + return json.dumps(text, ensure_ascii=False) + return text + + +def arxiv_id(entry_id: str) -> str: + raw = entry_id.rsplit("/", 1)[-1] + return raw.split("v", 1)[0] + + +def date_window_query(raw_query: str, from_date: str, to_date: str) -> str: + start = from_date.replace("-", "") + "0000" + end = to_date.replace("-", "") + "2359" + return f"({raw_query}) AND submittedDate:[{start} TO {end}]" + + +def fetch_query( + raw_query: str, + from_date: str, + to_date: str, + start: int, + max_results: int, + retries: int, + retry_sleep: float, +) -> str: + params = { + "search_query": date_window_query(raw_query, from_date, to_date), + "start": start, + "max_results": max_results, + "sortBy": "submittedDate", + "sortOrder": "descending", + } + url = f"{API_URL}?{urllib.parse.urlencode(params)}" + request = urllib.request.Request(url, headers={"User-Agent": "agent-kb-arxiv-collector/0.1"}) + for attempt in range(retries + 1): + try: + with urllib.request.urlopen(request, timeout=30) as response: + return response.read().decode("utf-8", "replace") + except urllib.error.HTTPError as exc: + if exc.code != 429 or attempt >= retries: + raise + time.sleep(retry_sleep * (attempt + 1)) + except urllib.error.URLError: + if attempt >= retries: + raise + time.sleep(retry_sleep * (attempt + 1)) + raise RuntimeError("unreachable fetch retry state") + + +def manifest_record(record: dict, topics: list[str], score: int, relevance: str, matched_queries: set[str]) -> dict: + sorted_queries = sorted(matched_queries) + return { + "id": record["id"], + "arxiv_id": record["id"], + "source": "arxiv", + "source_id": f"arxiv:{record['id']}", + "title": record["title"], + "url": record["url"], + "pdf_url": f"https://arxiv.org/pdf/{record['id']}", + "published": record["published"], + "updated": record["updated"], + "authors": record["authors"], + "categories": record["categories"], + "topics": topics, + "score": score, + "relevance": relevance, + "primary_query": sorted_queries[0] if sorted_queries else "", + "matched_queries": sorted_queries, + } + + +def parse_feed(xml_text: str) -> list[dict]: + root = ET.fromstring(xml_text) + records = [] + for entry in root.findall("atom:entry", NS): + entry_id = entry.findtext("atom:id", default="", namespaces=NS) + title = normalize_space(entry.findtext("atom:title", default="", namespaces=NS)) + summary = normalize_space(entry.findtext("atom:summary", default="", namespaces=NS)) + published = entry.findtext("atom:published", default="", namespaces=NS)[:10] + updated = entry.findtext("atom:updated", default="", namespaces=NS)[:10] + authors = [ + normalize_space(author.findtext("atom:name", default="", namespaces=NS)) + for author in entry.findall("atom:author", NS) + ] + categories = [ + category.attrib.get("term", "") + for category in entry.findall("atom:category", NS) + if category.attrib.get("term") + ] + records.append( + { + "id": arxiv_id(entry_id), + "url": f"https://arxiv.org/abs/{arxiv_id(entry_id)}", + "title": title, + "summary": summary, + "published": published, + "updated": updated, + "authors": authors, + "categories": categories, + } + ) + return records + + +def classify(record: dict, matched_queries: set[str]) -> tuple[list[str], int, str]: + text = f"{record['title']} {record['summary']}".lower() + title = record["title"].lower() + topics = [] + score = 0 + + if "agent" in title or "agentic" in title: + score += 4 + elif "agent" in text or "agentic" in text: + score += 2 + + if "llm" in text or "large language model" in text or "language model" in text: + score += 2 + + for term in HIGH_SIGNAL_TERMS: + if term in title: + score += 3 + elif term in text: + score += 1 + + for topic, keywords in TOPIC_RULES: + if any(keyword in text for keyword in keywords): + topics.append(topic) + score += 1 + + if matched_queries: + score += min(len(matched_queries), 4) + + if not topics and ("agent" in text or "agentic" in text): + topics.append("agent") + + if score >= 13: + relevance = "high" + elif score >= 7: + relevance = "medium" + else: + relevance = "low" + + return sorted(set(topics)), score, relevance + + +def existing_arxiv_ids() -> set[str]: + ids = set() + for path in (ROOT / "papers" / "items").glob("*.md"): + text = path.read_text(encoding="utf-8") + for match in re.findall(r"https://arxiv\.org/(?:abs|html)/([0-9]+\.[0-9]+)", text): + ids.add(match) + return ids + + +def item_path(record: dict) -> Path: + year = (record.get("published") or "0000")[:4] + title_slug = slugify(record["title"]) + id_slug = record["id"].replace(".", "-") + return ROOT / "papers" / "items" / f"{year}-{id_slug}-{title_slug}.md" + + +def write_item(record: dict, topics: list[str], score: int, relevance: str, matched_queries: set[str], today: str) -> Path: + path = item_path(record) + authors = ", ".join(record["authors"][:8]) + if len(record["authors"]) > 8: + authors += ", et al." + categories = ", ".join(record["categories"]) + query_list = ", ".join(sorted(matched_queries)) + topic_block = "\n".join(f" - {topic}" for topic in topics) or " - agent" + category_block = "\n".join(f" - {category}" for category in record["categories"]) or " -" + content = f"""# Paper: {record['title']} + +--- +type: paper +title: {yaml_value(record['title'])} +authors: {yaml_value(authors)} +year: {yaml_value((record.get('published') or '')[:4])} +venue: arXiv +url: {record['url']} +code_url: +source: arxiv +collected_at: {today} +published_at: {record.get('published') or ''} +updated_at: {record.get('updated') or ''} +status: queued +relevance: {relevance} +topics: +{topic_block} +methods: + - +benchmarks: + - +models: + - +datasets: +{category_block} +related_concepts: + - +related_jobs: + - +related_experiments: + - +related_projects: + - +collection_score: {score} +collection_queries: {yaml_value(query_list)} +--- + +## One-line Takeaway + +Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim. + +## Why Collected + +- matched queries: {query_list or 'agent-related arXiv sweep'} +- inferred topics: {', '.join(topics) or 'agent'} +- arXiv categories: {categories or 'unknown'} +- collection score: {score} + +## Review Checklist + +- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows? +- Does it include a benchmark, dataset, code, or reproducible experimental setup? +- Should it be promoted from `queued` to `skimmed` or `summarized`? + +## Links + +- arXiv: {record['url']} +""" + path.write_text(content, encoding="utf-8") + return path + + +def main() -> int: + parser = argparse.ArgumentParser() + parser.add_argument("--from-date", default="2025-07-08") + parser.add_argument("--to-date", default=date.today().isoformat()) + parser.add_argument("--per-query", type=int, default=120) + parser.add_argument("--page-size", type=int, default=60) + parser.add_argument("--max-items", type=int, default=180) + parser.add_argument("--min-score", type=int, default=6) + parser.add_argument("--sleep", type=float, default=1.0) + parser.add_argument("--retries", type=int, default=4) + parser.add_argument("--retry-sleep", type=float, default=10.0) + parser.add_argument("--dry-run", action="store_true") + args = parser.parse_args() + + by_id: dict[str, dict] = {} + matches: dict[str, set[str]] = defaultdict(set) + + for label, raw_query in QUERIES: + fetched = 0 + start = 0 + while fetched < args.per_query: + page_size = min(args.page_size, args.per_query - fetched) + xml_text = fetch_query( + raw_query, + args.from_date, + args.to_date, + start, + page_size, + args.retries, + args.retry_sleep, + ) + records = parse_feed(xml_text) + if not records: + break + for record in records: + by_id.setdefault(record["id"], record) + matches[record["id"]].add(label) + fetched += len(records) + start += len(records) + if len(records) < page_size: + break + time.sleep(args.sleep) + + existing = existing_arxiv_ids() + scored = [] + all_candidates = [] + for record_id, record in by_id.items(): + topics, score, relevance = classify(record, matches[record_id]) + all_candidates.append(manifest_record(record, topics, score, relevance, matches[record["id"]])) + if score < args.min_score: + continue + scored.append((score, relevance, topics, record)) + + scored.sort(key=lambda item: (item[0], item[3].get("published") or ""), reverse=True) + selected = scored[: args.max_items] + + data_dir = ROOT / "data" + data_dir.mkdir(exist_ok=True) + manifest = [] + written = [] + skipped_existing = 0 + today = date.today().isoformat() + for score, relevance, topics, record in selected: + manifest.append(manifest_record(record, topics, score, relevance, matches[record["id"]])) + if record["id"] in existing: + skipped_existing += 1 + continue + if not args.dry_run: + path = write_item(record, topics, score, relevance, matches[record["id"]], today) + written.append(str(path.relative_to(ROOT))) + + all_candidates.sort(key=lambda item: (item["score"], item["published"]), reverse=True) + candidates_path = data_dir / f"arxiv-agent-candidates-{args.from_date}-to-{args.to_date}.json" + candidates_path.write_text(json.dumps(all_candidates, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + manifest_path = data_dir / f"arxiv-agent-papers-{args.from_date}-to-{args.to_date}.json" + manifest_path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + summary = { + "from_date": args.from_date, + "to_date": args.to_date, + "queries": [label for label, _ in QUERIES], + "unique_seen": len(by_id), + "selected": len(selected), + "written": len(written), + "skipped_existing": skipped_existing, + "candidates_manifest": str(candidates_path.relative_to(ROOT)), + "manifest": str(manifest_path.relative_to(ROOT)), + "written_paths": written, + } + print(json.dumps(summary, ensure_ascii=False, indent=2)) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/collection/promote_arxiv_manifest.py b/tools/collection/promote_arxiv_manifest.py new file mode 100644 index 0000000..5a4730f --- /dev/null +++ b/tools/collection/promote_arxiv_manifest.py @@ -0,0 +1,101 @@ +#!/usr/bin/env python3 +"""Promote arXiv manifest records into paper item notes.""" + +from __future__ import annotations + +import argparse +import json +from datetime import date +from pathlib import Path + +from collect_arxiv import ROOT, existing_arxiv_ids, write_item + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser() + parser.add_argument("manifest", help="Path to an arXiv candidate manifest JSON file.") + parser.add_argument("--min-score", type=int, default=13) + parser.add_argument("--relevance", default="high") + parser.add_argument("--max-items", type=int, default=0, help="0 means no cap.") + parser.add_argument("--output-manifest", help="Optional JSON file for promoted records.") + parser.add_argument("--dry-run", action="store_true") + return parser.parse_args() + + +def normalize_record(record: dict) -> dict: + arxiv_id = record.get("arxiv_id") or record["id"] + return { + "id": arxiv_id, + "url": record.get("url") or f"https://arxiv.org/abs/{arxiv_id}", + "title": record["title"], + "published": record.get("published") or "", + "updated": record.get("updated") or "", + "authors": record.get("authors") or [], + "categories": record.get("categories") or [], + } + + +def main() -> int: + args = parse_args() + manifest_path = Path(args.manifest) + if not manifest_path.is_absolute(): + manifest_path = ROOT / manifest_path + + records = json.loads(manifest_path.read_text(encoding="utf-8")) + selected = [ + record + for record in records + if int(record.get("score") or 0) >= args.min_score + and (not args.relevance or record.get("relevance") == args.relevance) + ] + selected.sort(key=lambda item: (int(item.get("score") or 0), item.get("published") or ""), reverse=True) + if args.max_items: + selected = selected[: args.max_items] + + existing = existing_arxiv_ids() + today = date.today().isoformat() + written = [] + skipped_existing = 0 + for record in selected: + arxiv_id = record.get("arxiv_id") or record["id"] + if arxiv_id in existing: + skipped_existing += 1 + continue + if not args.dry_run: + path = write_item( + normalize_record(record), + record.get("topics") or ["agent"], + int(record.get("score") or 0), + record.get("relevance") or "unknown", + set(record.get("matched_queries") or []), + today, + ) + written.append(str(path.relative_to(ROOT))) + existing.add(arxiv_id) + + if args.output_manifest and not args.dry_run: + output_path = Path(args.output_manifest) + if not output_path.is_absolute(): + output_path = ROOT / output_path + output_path.parent.mkdir(parents=True, exist_ok=True) + output_path.write_text(json.dumps(selected, ensure_ascii=False, indent=2) + "\n", encoding="utf-8") + + print( + json.dumps( + { + "manifest": str(manifest_path.relative_to(ROOT)), + "selected": len(selected), + "written": len(written), + "skipped_existing": skipped_existing, + "output_manifest": args.output_manifest or "", + "written_paths": written, + }, + ensure_ascii=False, + indent=2, + ) + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main())