[ { "collection": "jobs", "path": "jobs/items/2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing.md", "title": "2027AIDU-智能体算法工程师", "type": "job", "meta": { "type": "job", "company": "百度", "role": "2027AIDU-智能体算法工程师", "location": "北京市", "source_url": "https://talent.baidu.com/jobs/list?projectType=3&recruitType=GRADUATE", "source_name": "百度校园招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": "2026-05-12", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "J99969", "previous_snapshot": [], "status": "active", "level": "graduate", "team": "AIDU项目", "business_area": "search / conversation / office / data platform / entertainment", "employment_type": "campus", "salary": [], "skills": [ "agent", "planning", "tool-use", "memory", "multi-agent", "rag", "agent-evaluation" ], "topics": [ "agent-architecture", "evaluation", "memory" ], "models": [], "related_papers": [ "2025-vijayvargiya-openagentsafety", "2026-memory-agent-survey" ], "related_experiments": [], "related_projects": [], "relevance": "high" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-baidu-aidu-agent-fullstack-engineer-beijing.md", "title": "2027AIDU-Agent应用全栈工程师", "type": "job", "meta": { "type": "job", "company": "百度", "role": "2027AIDU-Agent应用全栈工程师", "location": "北京市", "source_url": "https://talent.baidu.com/jobs/list?projectType=3&recruitType=GRADUATE", "source_name": "百度校园招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": "2026-05-12", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "J99974", "previous_snapshot": [], "status": "active", "level": "graduate", "team": "AIDU项目", "business_area": "search / healthcare / enterprise service / data analysis / office automation", "employment_type": "campus", "salary": [], "skills": [ "agent", "planning", "tool-use", "function-calling", "memory", "reasoning", "state-management", "multi-agent", "rag", "agent-evaluation" ], "topics": [ "workflow-agent", "agent-architecture", "evaluation" ], "models": [], "related_papers": [ "2026-dialogue-swebench", "2026-swe-evo" ], "related_experiments": [], "related_projects": [], "relevance": "high" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-baidu-aidu-llm-infra-engineer-beijing.md", "title": "2027AIDU-大模型Infra工程师", "type": "job", "meta": { "type": "job", "company": "百度", "role": "2027AIDU-大模型Infra工程师", "location": "北京市", "source_url": "https://talent.baidu.com/jobs/list?projectType=3&recruitType=GRADUATE", "source_name": "百度校园招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": "2026-05-12", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "J99967", "previous_snapshot": [], "status": "active", "level": "graduate", "team": "AIDU项目", "business_area": "model infrastructure", "employment_type": "campus", "salary": [], "skills": [ "inference", "serving", "distributed-training", "gpu", "model-compression", "cloud" ], "topics": [ "llm-infra", "inference" ], "models": [], "related_papers": [], "related_experiments": [], "related_projects": [], "relevance": "medium" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-bytedance-seed-llm-agent-research-engineer.md", "title": "大模型Agent研究工程师-Seed", "type": "job", "meta": { "type": "job", "company": "字节 Seed", "role": "大模型Agent研究工程师-Seed", "location": "unknown", "source_url": "https://jobs.bytedance.com/experienced/position/7628902314323806469/detail", "source_name": "字节跳动招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": [], "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "7628902314323806469", "previous_snapshot": [], "status": "unknown", "level": "experienced", "team": "Seed", "business_area": "agent research and engineering", "employment_type": "full-time", "salary": [], "skills": [ "agent", "memory", "context-engineering", "planning", "multi-agent", "llm-application" ], "topics": [ "harness", "memory", "context-compression" ], "models": [ "Seed" ], "related_papers": [ "2026-memory-agent-survey", "2026-evomembench" ], "related_experiments": [], "related_projects": [], "relevance": "high" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-deepseek-agent-hiring-wave-beijing-hangzhou.md", "title": "Agent 方向招聘组合", "type": "job", "meta": { "type": "job", "company": "DeepSeek", "role": "Agent 方向招聘组合", "location": "北京市 / 杭州市", "source_url": "https://hub.baai.ac.cn/view/53416", "source_name": "智源社区转载量子位", "source_quality": "secondary", "collected_at": "2026-07-08", "posted_at": "2026-03-27", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": [], "previous_snapshot": [], "status": "unknown", "level": "mixed", "team": "Agent / data evaluation / infra / product", "business_area": "search / creation / multimodal / personal assistant / workflow", "employment_type": "full-time / internship", "salary": [], "skills": [ "agent", "rl", "agent-evaluation", "tool-use", "function-calling", "memory", "multi-agent", "coding-agent", "data-quality", "inference" ], "topics": [ "agent-productization", "evaluation", "agent-infra" ], "models": [ "DeepSeek" ], "related_papers": [ "2025-vijayvargiya-openagentsafety", "2026-agentrx" ], "related_experiments": [], "related_projects": [], "relevance": "high" } }, { "collection": "jobs", "path": "jobs/items/2026-07-08-tencent-cloud-ai-agent-test-engineer.md", "title": "腾讯云-AI Agent测试工程师", "type": "job", "meta": { "type": "job", "company": "腾讯", "role": "腾讯云-AI Agent测试工程师", "location": "深圳市", "source_url": "https://careers.tencent.com/jobdesc.html?postId=2055555661177204736", "source_name": "腾讯招聘", "source_quality": "official", "collected_at": "2026-07-08", "posted_at": "2026-06-03", "first_seen_at": "2026-07-08", "last_checked_at": "2026-07-08", "snapshot_type": "new", "job_code": "2055555661177204736", "previous_snapshot": [], "status": "active", "level": "experienced", "team": "CSIG", "business_area": "cloud agent testing", "employment_type": "full-time", "salary": [], "skills": [ "agent", "agent-evaluation", "observability", "cloud", "kubernetes", "system-design" ], "topics": [ "agent-testing", "harness-engineering" ], "models": [], "related_papers": [ "2026-agentrx", "2025-vijayvargiya-openagentsafety" ], "related_experiments": [], "related_projects": [], "relevance": "medium" } }, { "collection": "papers", "path": "papers/items/2025-vijayvargiya-openagentsafety.md", "title": "OpenAgentSafety: A Comprehensive Framework for Evaluating Real-World AI Agent Safety", "type": "paper", "meta": { "type": "paper", "title": "OpenAgentSafety: A Comprehensive Framework for Evaluating Real-World AI Agent Safety", "authors": "Sanidhya Vijayvargiya, Aditya Bharat Soni, Xuhui Zhou, Zora Zhiruo Wang, Nouha Dziri, Graham Neubig, Maarten Sap", "year": "2025", "venue": "ICLR 2026 / IASEAI 2026", "url": "https://arxiv.org/abs/2507.06134", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "agent-evaluation", "agent-safety", "tool-use" ], "methods": [ "multi-turn-agent-evaluation", "rule-based-analysis", "llm-as-judge" ], "benchmarks": [ "OpenAgentSafety" ], "models": [], "datasets": [], "related_concepts": [ "guardrail", "human-in-the-loop" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing", "2026-07-08-tencent-cloud-ai-agent-test-engineer" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-agent-safety-benchmark-taxonomy.md", "title": "\"Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents\"", "authors": [], "year": "2026", "venue": [], "url": "https://arxiv.org/html/2605.16282v1", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "status": "queued", "relevance": "medium", "topics": [ "agent-safety", "agent-evaluation" ], "methods": [ "benchmark-taxonomy", "coverage-matrix" ], "benchmarks": [ "agent-safety-benchmarks" ], "models": [], "datasets": [], "related_concepts": [ "guardrail", "red-teaming" ], "related_jobs": [ "2026-07-08-tencent-cloud-ai-agent-test-engineer" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-king-dialogue-swebench.md", "title": "\"Dialogue-SWEBench: A Benchmark for Dialogue-Driven Coding Agents\"", "type": "paper", "meta": { "type": "paper", "title": "\"Dialogue-SWEBench: A Benchmark for Dialogue-Driven Coding Agents\"", "authors": "Brendan King, Jeffrey Flanigan", "year": "2026", "venue": [], "url": "https://arxiv.org/html/2606.13995v1", "code_url": "https://jlab-nlp.github.io/dialogue-swe-bench/", "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "coding-agent", "human-in-the-loop", "agent-evaluation" ], "methods": [ "user-simulator", "dialogue-benchmark", "schema-guided-agent" ], "benchmarks": [ "Dialogue-SWEBench" ], "models": [], "datasets": [ "SWE-Bench Verified" ], "related_concepts": [ "human-in-the-loop", "coding-agent" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-fullstack-engineer-beijing" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-memory-agent-survey.md", "title": "\"Memory for Autonomous LLM Agents: Mechanisms, Evaluation, and Emerging Frontiers\"", "type": "paper", "meta": { "type": "paper", "title": "\"Memory for Autonomous LLM Agents: Mechanisms, Evaluation, and Emerging Frontiers\"", "authors": [], "year": "2026", "venue": [], "url": "https://arxiv.org/html/2603.07670v1", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "memory", "agent-architecture", "agent-evaluation" ], "methods": [ "write-manage-read-loop", "retrieval-augmented-memory", "reflective-memory", "hierarchical-context" ], "benchmarks": [], "models": [], "datasets": [], "related_concepts": [ "memory", "context-engineering" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing", "2026-07-08-bytedance-seed-llm-agent-research-engineer" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-pham-swe-evo.md", "title": "\"SWE-EVO: Benchmarking Coding Agents in Long-Horizon Software Evolution Scenarios\"", "type": "paper", "meta": { "type": "paper", "title": "\"SWE-EVO: Benchmarking Coding Agents in Long-Horizon Software Evolution Scenarios\"", "authors": "Minh Vu Thai Pham, Tue Le, Dung Nguyen Manh, Huy Nhat Phan, Nghi D. Q. Bui", "year": "2026", "venue": [], "url": "https://arxiv.org/html/2512.18470v5", "code_url": "https://github.com/SWE-EVO/SWE-EVO", "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "coding-agent", "agent-evaluation" ], "methods": [ "long-horizon-software-evolution", "benchmark" ], "benchmarks": [ "SWE-EVO" ], "models": [ "OpenAI", "DeepSeek", "Zhipu", "Qwen", "Moonshot" ], "datasets": [ "release-notes" ], "related_concepts": [ "coding-agent", "long-horizon-task" ], "related_jobs": [ "2026-07-08-deepseek-agent-hiring-wave-beijing-hangzhou" ], "related_experiments": [], "related_projects": [] } }, { "collection": "papers", "path": "papers/items/2026-wang-evomembench.md", "title": "\"EvoMemBench: Benchmarking Agent Memory from a Self-Evolving Perspective\"", "type": "paper", "meta": { "type": "paper", "title": "\"EvoMemBench: Benchmarking Agent Memory from a Self-Evolving Perspective\"", "authors": "Yuyao Wang, Zhongjian Zhang, Mo Chi, Kaichi Yu, Yuhan Li, Miao Peng, Bing Tong, Chen Zhang, Yan Zhou, Jia Li", "year": "2026", "venue": [], "url": "https://arxiv.org/html/2605.18421", "code_url": [], "source": "arxiv", "collected_at": "2026-07-08", "status": "skimmed", "relevance": "high", "topics": [ "memory", "agent-evaluation" ], "methods": [ "memory-benchmark", "self-evolving-agent" ], "benchmarks": [ "EvoMemBench" ], "models": [], "datasets": [], "related_concepts": [ "memory", "procedural-memory" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing", "2026-07-08-bytedance-seed-llm-agent-research-engineer" ], "related_experiments": [], "related_projects": [] } }, { "collection": "industry", "path": "industry/items/2026-07-08-bytedance-seed2-1-agent-productivity.md", "title": "\"Seed2.1 Officially Released: Advancing AI Productivity\"", "type": "industry", "meta": { "type": "industry", "company": "ByteDance Seed", "team": "Seed", "title": "\"Seed2.1 Officially Released: Advancing AI Productivity\"", "url": "https://seed.bytedance.com/en/blog/seed2-1-officially-released-advancing-ai-productivity", "source_name": "ByteDance Seed", "source_type": "research-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent", "coding-agent", "gui-agent" ], "implementation_signals": [ "reinforcement-learning", "gui-tool-use", "mcp", "crowdsourced-evaluation", "internal-benchmark" ], "product_area": [ "productivity", "coding-agent", "office-agent" ], "models": [ "Seed2.1" ], "tools": [ "MCP" ], "benchmarks": [ "CreativeWork", "ProgramBench" ], "related_papers": [ "2026-pham-swe-evo" ], "related_jobs": [ "2026-07-08-bytedance-seed-llm-agent-research-engineer" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-google-agent-security-roadmap.md", "title": "Securing the future of AI agents", "type": "industry", "meta": { "type": "industry", "company": "Google DeepMind", "team": [], "title": "Securing the future of AI agents", "url": "https://deepmind.google/blog/securing-the-future-of-ai-agents/", "source_name": "Google DeepMind", "source_type": "technical-report", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent-safety", "agent-security", "governance" ], "implementation_signals": [ "defense-in-depth", "sandboxing", "permission-control", "prompt-injection-resistance" ], "product_area": [ "internal-agents", "enterprise-agent" ], "models": [], "tools": [], "benchmarks": [], "related_papers": [ "2025-vijayvargiya-openagentsafety", "2026-agent-safety-benchmark-taxonomy" ], "related_jobs": [ "2026-07-08-tencent-cloud-ai-agent-test-engineer" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-google-gemini-computer-use.md", "title": "Introducing computer use in Gemini 3.5 Flash", "type": "industry", "meta": { "type": "industry", "company": "Google DeepMind", "team": "Gemini", "title": "Introducing computer use in Gemini 3.5 Flash", "url": "https://deepmind.google/blog/introducing-computer-use-in-gemini-3-5-flash/", "source_name": "Google DeepMind / Google Blog", "source_type": "product-blog", "source_quality": "official", "published_at": "2026-06-24", "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "computer-use", "agent", "tool-use" ], "implementation_signals": [ "browser-control", "gui-control", "safety", "human-in-the-loop" ], "product_area": [ "enterprise-agent", "automation" ], "models": [ "Gemini 3.5 Flash" ], "tools": [ "Gemini API", "Gemini Enterprise Agent Platform" ], "benchmarks": [], "related_papers": [ "2025-vijayvargiya-openagentsafety" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-fullstack-engineer-beijing" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-microsoft-agentrx.md", "title": "\"Systematic debugging for AI agents: Introducing the AgentRx framework\"", "type": "industry", "meta": { "type": "industry", "company": "Microsoft", "team": "Microsoft Research", "title": "\"Systematic debugging for AI agents: Introducing the AgentRx framework\"", "url": "https://www.microsoft.com/en-us/research/blog/systematic-debugging-for-ai-agents-introducing-the-agentrx-framework/", "source_name": "Microsoft Research", "source_type": "research-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent-evaluation", "observability", "debugging" ], "implementation_signals": [ "trajectory-normalization", "constraint-synthesis", "constraint-checking", "failure-taxonomy" ], "product_area": [ "agent-debugging" ], "models": [], "tools": [ "AgentRx" ], "benchmarks": [ "AgentRx Benchmark" ], "related_papers": [ "2025-vijayvargiya-openagentsafety" ], "related_jobs": [ "2026-07-08-tencent-cloud-ai-agent-test-engineer" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-microsoft-skillopt.md", "title": "\"SkillOpt: Agent skills as trainable parameters\"", "type": "industry", "meta": { "type": "industry", "company": "Microsoft", "team": "Microsoft Research", "title": "\"SkillOpt: Agent skills as trainable parameters\"", "url": "https://www.microsoft.com/en-us/research/blog/skillopt-agent-skills-as-trainable-parameters/", "source_name": "Microsoft Research", "source_type": "research-blog", "source_quality": "official", "published_at": "2026-06-30", "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent", "prompt-optimization", "skill-learning" ], "implementation_signals": [ "skill-files", "validation-gating", "bounded-edits", "eval-loop" ], "product_area": [ "agent-framework" ], "models": [], "tools": [], "benchmarks": [ "six-benchmark-evaluation" ], "related_papers": [], "related_jobs": [ "2026-07-08-deepseek-agent-hiring-wave-beijing-hangzhou" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-microsoft-state-bench.md", "title": "\"Introducing STATE-Bench: a benchmark for AI agent memory\"", "type": "industry", "meta": { "type": "industry", "company": "Microsoft", "team": "Microsoft Open Source", "title": "\"Introducing STATE-Bench: a benchmark for AI agent memory\"", "url": "https://opensource.microsoft.com/blog/2026/05/19/introducing-state-bench-a-benchmark-for-ai-agent-memory/", "source_name": "Microsoft Open Source Blog", "source_type": "benchmark", "source_quality": "official", "published_at": "2026-05-19", "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "memory", "agent-evaluation", "enterprise-ai" ], "implementation_signals": [ "stateful-environment", "user-simulator", "deterministic-assertions", "bring-your-own-memory" ], "product_area": [ "customer-support", "travel", "shopping" ], "models": [], "tools": [ "STATE-Bench" ], "benchmarks": [ "STATE-Bench" ], "related_papers": [ "2026-memory-agent-survey", "2026-wang-evomembench" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-algorithm-engineer-beijing" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-openai-agents-transforming-work.md", "title": "How agents are transforming work", "type": "industry", "meta": { "type": "industry", "company": "OpenAI", "team": [], "title": "How agents are transforming work", "url": "https://openai.com/index/how-agents-are-transforming-work/", "source_name": "OpenAI", "source_type": "research-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent", "enterprise-ai", "coding-agent" ], "implementation_signals": [ "adoption-metrics", "workflow-change" ], "product_area": [ "coding-agent", "enterprise-work" ], "models": [ "Codex" ], "tools": [], "benchmarks": [], "related_papers": [], "related_jobs": [ "2026-07-08-deepseek-agent-hiring-wave-beijing-hangzhou" ], "related_experiments": [], "related_projects": [], "evidence_level": "medium", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-openai-in-house-data-agent.md", "title": "Inside OpenAI's in-house data agent", "type": "industry", "meta": { "type": "industry", "company": "OpenAI", "team": "data / engineering", "title": "Inside OpenAI's in-house data agent", "url": "https://openai.com/index/inside-our-in-house-data-agent/", "source_name": "OpenAI", "source_type": "engineering-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "analyzed", "topics": [ "agent", "data-agent", "enterprise-ai" ], "implementation_signals": [ "memory", "rag", "mcp", "eval", "permissions" ], "product_area": [ "data-analysis", "internal-tools" ], "models": [ "GPT-5.2", "Codex" ], "tools": [ "Evals API", "Embeddings API", "MCP" ], "benchmarks": [], "related_papers": [ "2026-memory-agent-survey" ], "related_jobs": [ "2026-07-08-baidu-aidu-agent-fullstack-engineer-beijing" ], "related_experiments": [], "related_projects": [], "evidence_level": "high", "relevance": "high" } }, { "collection": "industry", "path": "industry/items/2026-07-08-qwen-agentworld.md", "title": "\"Qwen-AgentWorld: Language World Models for General Agents\"", "type": "industry", "meta": { "type": "industry", "company": "Alibaba Qwen", "team": "Qwen", "title": "\"Qwen-AgentWorld: Language World Models for General Agents\"", "url": "https://qwen.ai/blog?id=qwen-agentworld", "source_name": "Qwen Blog", "source_type": "research-blog", "source_quality": "official", "published_at": [], "collected_at": "2026-07-08", "status": "queued", "topics": [ "agent", "world-model", "evaluation" ], "implementation_signals": [ "simulated-environment", "world-model", "multi-domain-agent" ], "product_area": [ "general-agent" ], "models": [ "Qwen" ], "tools": [], "benchmarks": [], "related_papers": [], "related_jobs": [], "related_experiments": [], "related_projects": [], "evidence_level": "medium", "relevance": "high" } } ]