Files
2026-07-27 16:28:23 +08:00

675 lines
24 KiB
Python

#!/usr/bin/env python3
"""Collect recent arXiv papers related to LLM/AI agents."""
from __future__ import annotations
import argparse
import http.client
import json
import re
import sys
import time
import urllib.parse
import urllib.request
import urllib.error
import xml.etree.ElementTree as ET
from collections import defaultdict
from datetime import date, datetime
from html.parser import HTMLParser
from pathlib import Path
ROOT = Path(__file__).resolve().parents[2]
API_URL = "https://export.arxiv.org/api/query"
SEARCH_URL = "https://arxiv.org/search/"
NS = {
"atom": "http://www.w3.org/2005/Atom",
"arxiv": "http://arxiv.org/schemas/atom",
}
QUERIES = [
("llm-agent", 'all:"LLM agent" OR all:"LLM agents"'),
("language-agent", 'all:"language agent" OR all:"language agents"'),
("ai-agent", 'all:"AI agent" OR all:"AI agents"'),
("agentic-ai", 'all:"agentic AI" OR all:"agentic workflow"'),
("agent-evaluation", 'all:"agent evaluation" OR all:"agent benchmark" OR all:"agentic benchmark"'),
("agent-memory", 'all:"agent memory" OR all:"memory agent" OR all:"memory agents"'),
("tool-use", 'all:"tool use" AND all:"agent"'),
("function-calling", 'all:"function calling" AND all:"agent"'),
("coding-agent", 'all:"coding agent" OR all:"software engineering agent" OR all:"SWE-bench"'),
("web-gui-agent", 'all:"web agent" OR all:"browser agent" OR all:"GUI agent" OR all:"computer use"'),
("multi-agent-llm", 'all:"multi-agent" AND (all:"LLM" OR all:"large language model")'),
("agent-safety", 'all:"agent safety" OR all:"AI agent safety" OR all:"agent security"'),
("rag-agent", 'all:"RAG" AND all:"agent"'),
("planning-agent", 'all:"planning" AND all:"LLM agent"'),
("autonomous-agent-llm", 'all:"autonomous agent" AND (all:"LLM" OR all:"large language model")'),
]
SEARCH_QUERIES = {
"llm-agent": '"LLM agent" OR "LLM agents"',
"language-agent": '"language agent" OR "language agents"',
"ai-agent": '"AI agent" OR "AI agents"',
"agentic-ai": '"agentic AI" OR "agentic workflow"',
"agent-evaluation": '"agent evaluation" OR "agent benchmark" OR "agentic benchmark"',
"agent-memory": '"agent memory" OR "memory agent"',
"tool-use": '"tool use" agent',
"function-calling": '"function calling" agent',
"coding-agent": '"coding agent" OR "software engineering agent" OR SWE-bench',
"web-gui-agent": '"web agent" OR "browser agent" OR "GUI agent" OR "computer use"',
"multi-agent-llm": '"multi-agent" LLM',
"agent-safety": '"agent safety" OR "agent security"',
"rag-agent": "RAG agent",
"planning-agent": 'planning "LLM agent"',
"autonomous-agent-llm": '"autonomous agent" LLM',
}
TOPIC_RULES = [
("agent-evaluation", ("evaluation", "benchmark", "eval", "metric", "leaderboard", "assessment")),
("memory", ("memory", "remember", "stateful", "long-term", "episodic")),
("tool-use", ("tool", "function calling", "api", "mcp", "action")),
("coding-agent", ("coding", "software engineering", "swe-bench", "repository", "program repair", "code agent")),
("computer-use", ("computer use", "gui", "browser", "web agent", "mobile", "desktop")),
("multi-agent", ("multi-agent", "multiagent", "agent society", "collaboration", "debate")),
("agent-safety", ("safety", "security", "risk", "prompt injection", "alignment", "guardrail")),
("rag", ("retrieval", "rag", "knowledge base", "grounding")),
("planning", ("planning", "planner", "plan", "long-horizon", "task decomposition")),
("reasoning", ("reasoning", "reflection", "self-improvement", "verifier")),
("workflow-agent", ("workflow", "enterprise", "office", "productivity", "automation")),
("embodied-agent", ("embodied", "robot", "robotic", "navigation")),
("world-model", ("world model", "simulation", "environment model")),
]
HIGH_SIGNAL_TERMS = (
"llm agent",
"language agent",
"ai agent",
"agentic",
"tool use",
"memory",
"benchmark",
"evaluation",
"coding agent",
"swe-bench",
"computer use",
"gui agent",
"multi-agent",
"agent safety",
)
def normalize_space(value: str) -> str:
return re.sub(r"\s+", " ", value).strip()
def slugify(text: str) -> str:
text = text.lower()
text = re.sub(r"[^a-z0-9]+", "-", text)
return re.sub(r"-+", "-", text).strip("-")[:80] or "untitled"
def yaml_value(value: str | int | None) -> str:
if value is None:
return ""
text = str(value).replace("\n", " ").strip()
if not text:
return ""
if any(char in text for char in [":", "#", "[", "]", "{", "}", "\"", "'"]):
return json.dumps(text, ensure_ascii=False)
return text
def arxiv_id(entry_id: str) -> str:
raw = entry_id.rsplit("/", 1)[-1]
return raw.split("v", 1)[0]
def date_window_query(raw_query: str, from_date: str, to_date: str) -> str:
start = from_date.replace("-", "") + "0000"
end = to_date.replace("-", "") + "2359"
return f"({raw_query}) AND submittedDate:[{start} TO {end}]"
def fetch_query(
raw_query: str,
from_date: str,
to_date: str,
start: int,
max_results: int,
retries: int,
retry_sleep: float,
request_timeout: float,
) -> str:
params = {
"search_query": date_window_query(raw_query, from_date, to_date),
"start": start,
"max_results": max_results,
"sortBy": "submittedDate",
"sortOrder": "descending",
}
url = f"{API_URL}?{urllib.parse.urlencode(params)}"
request = urllib.request.Request(url, headers={"User-Agent": "agent-kb-arxiv-collector/0.1"})
for attempt in range(retries + 1):
try:
with urllib.request.urlopen(request, timeout=request_timeout) as response:
return response.read().decode("utf-8", "replace")
except urllib.error.HTTPError as exc:
if exc.code not in {429, 500, 502, 503, 504} or attempt >= retries:
raise
print(
f"arXiv returned HTTP {exc.code}; retrying request "
f"{attempt + 1}/{retries} after backoff",
file=sys.stderr,
flush=True,
)
time.sleep(retry_sleep * (attempt + 1))
except (urllib.error.URLError, TimeoutError, http.client.HTTPException) as exc:
if attempt >= retries:
raise
print(
f"arXiv request failed with {type(exc).__name__}; retrying "
f"{attempt + 1}/{retries} after backoff",
file=sys.stderr,
flush=True,
)
time.sleep(retry_sleep * (attempt + 1))
raise RuntimeError("unreachable fetch retry state")
class ArxivSearchParser(HTMLParser):
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.records: list[dict] = []
self.current: dict | None = None
self.capture = ""
self.capture_tag = ""
self.capture_depth = 0
self.buffer: list[str] = []
@staticmethod
def classes(attrs: list[tuple[str, str | None]]) -> set[str]:
values = dict(attrs).get("class") or ""
return set(values.split())
def start_capture(self, field: str, tag: str) -> None:
self.capture = field
self.capture_tag = tag
self.capture_depth = 1
self.buffer = []
def finish_capture(self) -> None:
if self.current is None:
return
value = normalize_space("".join(self.buffer))
if self.capture == "authors":
value = re.sub(r"^Authors:\s*", "", value)
self.current["authors"] = [part.strip() for part in value.split(",") if part.strip()]
elif self.capture == "summary":
self.current["summary"] = re.sub(r"(?:△|▽)?\s*Less\s*$", "", value).strip()
elif self.capture == "category":
if value:
self.current["categories"].append(value)
else:
self.current[self.capture] = value
self.capture = ""
self.capture_tag = ""
self.capture_depth = 0
self.buffer = []
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
classes = self.classes(attrs)
if tag == "li" and "arxiv-result" in classes:
self.current = {
"id": "",
"url": "",
"title": "",
"summary": "",
"submitted": "",
"authors": [],
"categories": [],
}
return
if self.current is None:
return
if self.capture:
self.capture_depth += 1
elif tag == "p" and "title" in classes:
self.start_capture("title", tag)
elif tag == "p" and "authors" in classes:
self.start_capture("authors", tag)
elif tag == "span" and "abstract-full" in classes:
self.start_capture("summary", tag)
elif tag == "span" and "tag" in classes and "is-link" in classes:
self.start_capture("category", tag)
elif tag == "p" and "is-size-7" in classes:
self.start_capture("submitted", tag)
href = dict(attrs).get("href") or ""
match = re.fullmatch(r"https://arxiv\.org/abs/([0-9]+\.[0-9]+)(?:v[0-9]+)?", href)
if tag == "a" and match and not self.current["id"]:
self.current["id"] = match.group(1)
self.current["url"] = f"https://arxiv.org/abs/{match.group(1)}"
def handle_endtag(self, tag: str) -> None:
if self.current is None:
return
if self.capture:
self.capture_depth -= 1
if self.capture_depth == 0 and tag == self.capture_tag:
self.finish_capture()
if tag == "li":
record = self.current
self.current = None
if record["id"] and record["title"]:
submitted_match = re.search(
r"Submitted\s+([0-9]{1,2}\s+[A-Za-z]+,\s+[0-9]{4})",
record.pop("submitted", ""),
)
if submitted_match:
published = datetime.strptime(submitted_match.group(1), "%d %B, %Y").date().isoformat()
record["published"] = published
record["updated"] = published
self.records.append(record)
def handle_data(self, data: str) -> None:
if self.capture:
self.buffer.append(data)
def fetch_search_page(
query: str,
start: int,
max_results: int,
retries: int,
retry_sleep: float,
request_timeout: float,
) -> str:
params = {
"query": query,
"searchtype": "all",
"abstracts": "show",
"order": "-announced_date_first",
"size": min(max_results, 200),
"start": start,
}
url = f"{SEARCH_URL}?{urllib.parse.urlencode(params)}"
request = urllib.request.Request(url, headers={"User-Agent": "agent-kb-arxiv-collector/0.2"})
for attempt in range(retries + 1):
try:
with urllib.request.urlopen(request, timeout=request_timeout) as response:
return response.read().decode("utf-8", "replace")
except urllib.error.HTTPError as exc:
if exc.code not in {429, 500, 502, 503, 504} or attempt >= retries:
raise
print(
f"arXiv search returned HTTP {exc.code}; retrying request "
f"{attempt + 1}/{retries} after backoff",
file=sys.stderr,
flush=True,
)
except (urllib.error.URLError, TimeoutError, http.client.HTTPException) as exc:
if attempt >= retries:
raise
print(
f"arXiv search failed with {type(exc).__name__}; retrying "
f"{attempt + 1}/{retries} after backoff",
file=sys.stderr,
flush=True,
)
time.sleep(retry_sleep * (attempt + 1))
raise RuntimeError("unreachable search retry state")
def fetch_search_query(
query: str,
from_date: str,
to_date: str,
per_query: int,
page_size: int,
sleep_seconds: float,
retries: int,
retry_sleep: float,
request_timeout: float,
) -> list[dict]:
records: list[dict] = []
start = 0
while start < per_query:
requested = min(page_size, per_query - start, 200)
html_text = fetch_search_page(
query,
start,
requested,
retries,
retry_sleep,
request_timeout,
)
parser = ArxivSearchParser()
parser.feed(html_text)
page_records = parser.records
if not page_records:
break
records.extend(record for record in page_records if from_date <= record["published"] <= to_date)
print(
f"searched arXiv HTML: {start + len(page_records)} records, "
f"{len(records)} inside window",
file=sys.stderr,
flush=True,
)
start += len(page_records)
if len(page_records) < requested:
break
# Announced-date ordering can mix old cross-listed papers into a recent page,
# so the original submission date is not a safe early-stop signal.
time.sleep(sleep_seconds)
return records
def manifest_record(record: dict, topics: list[str], score: int, relevance: str, matched_queries: set[str]) -> dict:
sorted_queries = sorted(matched_queries)
return {
"id": record["id"],
"arxiv_id": record["id"],
"source": "arxiv",
"source_id": f"arxiv:{record['id']}",
"title": record["title"],
"url": record["url"],
"pdf_url": f"https://arxiv.org/pdf/{record['id']}",
"published": record["published"],
"updated": record["updated"],
"authors": record["authors"],
"categories": record["categories"],
"topics": topics,
"score": score,
"relevance": relevance,
"primary_query": sorted_queries[0] if sorted_queries else "",
"matched_queries": sorted_queries,
}
def parse_feed(xml_text: str) -> list[dict]:
root = ET.fromstring(xml_text)
records = []
for entry in root.findall("atom:entry", NS):
entry_id = entry.findtext("atom:id", default="", namespaces=NS)
title = normalize_space(entry.findtext("atom:title", default="", namespaces=NS))
summary = normalize_space(entry.findtext("atom:summary", default="", namespaces=NS))
published = entry.findtext("atom:published", default="", namespaces=NS)[:10]
updated = entry.findtext("atom:updated", default="", namespaces=NS)[:10]
authors = [
normalize_space(author.findtext("atom:name", default="", namespaces=NS))
for author in entry.findall("atom:author", NS)
]
categories = [
category.attrib.get("term", "")
for category in entry.findall("atom:category", NS)
if category.attrib.get("term")
]
records.append(
{
"id": arxiv_id(entry_id),
"url": f"https://arxiv.org/abs/{arxiv_id(entry_id)}",
"title": title,
"summary": summary,
"published": published,
"updated": updated,
"authors": authors,
"categories": categories,
}
)
return records
def classify(record: dict, matched_queries: set[str]) -> tuple[list[str], int, str]:
text = f"{record['title']} {record['summary']}".lower()
title = record["title"].lower()
topics = []
score = 0
if "agent" in title or "agentic" in title:
score += 4
elif "agent" in text or "agentic" in text:
score += 2
if "llm" in text or "large language model" in text or "language model" in text:
score += 2
for term in HIGH_SIGNAL_TERMS:
if term in title:
score += 3
elif term in text:
score += 1
for topic, keywords in TOPIC_RULES:
if any(keyword in text for keyword in keywords):
topics.append(topic)
score += 1
if matched_queries:
score += min(len(matched_queries), 4)
if not topics and ("agent" in text or "agentic" in text):
topics.append("agent")
if score >= 13:
relevance = "high"
elif score >= 7:
relevance = "medium"
else:
relevance = "low"
return sorted(set(topics)), score, relevance
def existing_arxiv_ids() -> set[str]:
ids = set()
for path in (ROOT / "papers" / "items").glob("*.md"):
text = path.read_text(encoding="utf-8")
for match in re.findall(r"https://arxiv\.org/(?:abs|html)/([0-9]+\.[0-9]+)", text):
ids.add(match)
return ids
def item_path(record: dict) -> Path:
year = (record.get("published") or "0000")[:4]
title_slug = slugify(record["title"])
id_slug = record["id"].replace(".", "-")
return ROOT / "papers" / "items" / f"{year}-{id_slug}-{title_slug}.md"
def write_item(record: dict, topics: list[str], score: int, relevance: str, matched_queries: set[str], today: str) -> Path:
path = item_path(record)
authors = ", ".join(record["authors"][:8])
if len(record["authors"]) > 8:
authors += ", et al."
categories = ", ".join(record["categories"])
query_list = ", ".join(sorted(matched_queries))
topic_block = "\n".join(f" - {topic}" for topic in topics) or " - agent"
category_block = "\n".join(f" - {category}" for category in record["categories"]) or " -"
content = f"""# Paper: {record['title']}
---
type: paper
title: {yaml_value(record['title'])}
authors: {yaml_value(authors)}
year: {yaml_value((record.get('published') or '')[:4])}
venue: arXiv
url: {record['url']}
code_url:
source: arxiv
collected_at: {today}
published_at: {record.get('published') or ''}
updated_at: {record.get('updated') or ''}
status: queued
relevance: {relevance}
topics:
{topic_block}
methods:
-
benchmarks:
-
models:
-
datasets:
{category_block}
related_concepts:
-
related_jobs:
-
related_experiments:
-
related_projects:
-
collection_score: {score}
collection_queries: {yaml_value(query_list)}
---
## One-line Takeaway
Auto-collected from arXiv because it matched the Agent collection queries. Needs human skim.
## Why Collected
- matched queries: {query_list or 'agent-related arXiv sweep'}
- inferred topics: {', '.join(topics) or 'agent'}
- arXiv categories: {categories or 'unknown'}
- collection score: {score}
## Review Checklist
- Does this paper directly inform Agent architecture, evaluation, memory, tools, safety, coding agents, GUI/browser agents, or multi-agent workflows?
- Does it include a benchmark, dataset, code, or reproducible experimental setup?
- Should it be promoted from `queued` to `skimmed` or `summarized`?
## Links
- arXiv: {record['url']}
"""
path.write_text(content, encoding="utf-8")
return path
def main() -> int:
parser = argparse.ArgumentParser()
parser.add_argument("--from-date", default="2025-07-08")
parser.add_argument("--to-date", default=date.today().isoformat())
parser.add_argument("--per-query", type=int, default=120)
parser.add_argument("--page-size", type=int, default=60)
parser.add_argument("--max-items", type=int, default=180)
parser.add_argument("--min-score", type=int, default=6)
parser.add_argument("--sleep", type=float, default=1.0)
parser.add_argument("--retries", type=int, default=4)
parser.add_argument("--retry-sleep", type=float, default=10.0)
parser.add_argument("--request-timeout", type=float, default=60.0)
parser.add_argument("--backend", choices=("api", "search-html"), default="api")
parser.add_argument("--dry-run", action="store_true")
args = parser.parse_args()
by_id: dict[str, dict] = {}
matches: dict[str, set[str]] = defaultdict(set)
for label, raw_query in QUERIES:
if args.backend == "search-html":
print(f"collecting {label} from arXiv search", file=sys.stderr, flush=True)
records = fetch_search_query(
SEARCH_QUERIES[label],
args.from_date,
args.to_date,
args.per_query,
args.page_size,
args.sleep,
args.retries,
args.retry_sleep,
args.request_timeout,
)
for record in records:
by_id.setdefault(record["id"], record)
matches[record["id"]].add(label)
continue
fetched = 0
start = 0
print(f"collecting {label}", file=sys.stderr, flush=True)
while fetched < args.per_query:
page_size = min(args.page_size, args.per_query - fetched)
xml_text = fetch_query(
raw_query,
args.from_date,
args.to_date,
start,
page_size,
args.retries,
args.retry_sleep,
args.request_timeout,
)
records = parse_feed(xml_text)
if not records:
break
for record in records:
by_id.setdefault(record["id"], record)
matches[record["id"]].add(label)
fetched += len(records)
start += len(records)
print(
f"collected {label}: {fetched} records",
file=sys.stderr,
flush=True,
)
if len(records) < page_size:
break
time.sleep(args.sleep)
existing = existing_arxiv_ids()
scored = []
all_candidates = []
for record_id, record in by_id.items():
topics, score, relevance = classify(record, matches[record_id])
all_candidates.append(manifest_record(record, topics, score, relevance, matches[record["id"]]))
if score < args.min_score:
continue
scored.append((score, relevance, topics, record))
scored.sort(key=lambda item: (item[0], item[3].get("published") or ""), reverse=True)
selected = scored[: args.max_items]
data_dir = ROOT / "data"
data_dir.mkdir(exist_ok=True)
manifest = []
written = []
skipped_existing = 0
today = date.today().isoformat()
for score, relevance, topics, record in selected:
manifest.append(manifest_record(record, topics, score, relevance, matches[record["id"]]))
if record["id"] in existing:
skipped_existing += 1
continue
if not args.dry_run:
path = write_item(record, topics, score, relevance, matches[record["id"]], today)
written.append(str(path.relative_to(ROOT)))
all_candidates.sort(key=lambda item: (item["score"], item["published"]), reverse=True)
candidates_path = data_dir / f"arxiv-agent-candidates-{args.from_date}-to-{args.to_date}.json"
candidates_path.write_text(json.dumps(all_candidates, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
manifest_path = data_dir / f"arxiv-agent-papers-{args.from_date}-to-{args.to_date}.json"
manifest_path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
summary = {
"from_date": args.from_date,
"to_date": args.to_date,
"queries": [label for label, _ in QUERIES],
"unique_seen": len(by_id),
"selected": len(selected),
"written": len(written),
"skipped_existing": skipped_existing,
"candidates_manifest": str(candidates_path.relative_to(ROOT)),
"manifest": str(manifest_path.relative_to(ROOT)),
"written_paths": written,
}
print(json.dumps(summary, ensure_ascii=False, indent=2))
return 0
if __name__ == "__main__":
raise SystemExit(main())