{
  "object_type": "arxiv_research_track",
  "id": "f5ec2268-4c6e-57d5-839c-d58e4abf6065",
  "slug": "ai-agents",
  "name": "AI Agents",
  "kind": "research_theme",
  "definition": {
    "description": "Research on autonomous and agentic AI systems, tool-using agents and agent-oriented evaluation, excluding papers already classified as multi-agent systems.",
    "exclude_tracks": [
      "multi-agent-systems"
    ],
    "featured_on_overview": true,
    "include_entities": [
      "Autonomous Agents",
      "Agentic AI",
      "ReAct",
      "Tool Use",
      "Function Calling",
      "AgentBench",
      "WebArena",
      "GAIA"
    ],
    "kind": "research_theme",
    "match_mode": "entity",
    "name": "AI Agents",
    "public_path": "agents",
    "slug": "ai-agents",
    "title_focus_semantics": "in_title=1 means at least one defining entity is present in the paper title."
  },
  "generated_at": "2026-09-04T01:00:05.757562Z",
  "counts": {
    "papers": 4948,
    "title_focus": 1309,
    "with_code": 772
  },
  "date_range": {
    "oldest": "2007-09-15",
    "newest": "2026-07-23"
  },
  "trend": [
    {
      "period": "2007",
      "paper_count": 1
    },
    {
      "period": "2011",
      "paper_count": 4
    },
    {
      "period": "2012",
      "paper_count": 3
    },
    {
      "period": "2013",
      "paper_count": 8
    },
    {
      "period": "2014",
      "paper_count": 6
    },
    {
      "period": "2015",
      "paper_count": 8
    },
    {
      "period": "2016",
      "paper_count": 18
    },
    {
      "period": "2017",
      "paper_count": 40
    },
    {
      "period": "2018",
      "paper_count": 77
    },
    {
      "period": "2019",
      "paper_count": 97
    },
    {
      "period": "2020",
      "paper_count": 93
    },
    {
      "period": "2021",
      "paper_count": 149
    },
    {
      "period": "2022",
      "paper_count": 149
    },
    {
      "period": "2023",
      "paper_count": 289
    },
    {
      "period": "2024",
      "paper_count": 610
    },
    {
      "period": "2025",
      "paper_count": 1574
    },
    {
      "period": "2026",
      "paper_count": 1822
    }
  ],
  "top_entities": [
    {
      "name": "RAG",
      "type": "method",
      "paper_count": 183,
      "title_mentions": 30
    },
    {
      "name": "Claude",
      "type": "model",
      "paper_count": 179,
      "title_mentions": 2
    },
    {
      "name": "VLM",
      "type": "method",
      "paper_count": 124,
      "title_mentions": 20
    },
    {
      "name": "Chain-of-Thought",
      "type": "method",
      "paper_count": 121,
      "title_mentions": 6
    },
    {
      "name": "GPT-4o",
      "type": "model",
      "paper_count": 110,
      "title_mentions": 0
    },
    {
      "name": "Gemini",
      "type": "model",
      "paper_count": 108,
      "title_mentions": 1
    },
    {
      "name": "LLaMA",
      "type": "model",
      "paper_count": 96,
      "title_mentions": 1
    },
    {
      "name": "SFT",
      "type": "method",
      "paper_count": 95,
      "title_mentions": 2
    },
    {
      "name": "Zero-Shot",
      "type": "method",
      "paper_count": 94,
      "title_mentions": 16
    },
    {
      "name": "GPT-4",
      "type": "model",
      "paper_count": 89,
      "title_mentions": 4
    },
    {
      "name": "Interpretability",
      "type": "method",
      "paper_count": 87,
      "title_mentions": 8
    },
    {
      "name": "Alignment",
      "type": "method",
      "paper_count": 81,
      "title_mentions": 18
    }
  ],
  "top_orgs": [
    {
      "org": "langchain-ai",
      "paper_count": 121,
      "repo_count": 12
    },
    {
      "org": "significant-gravitas",
      "paper_count": 86,
      "repo_count": 8
    },
    {
      "org": "microsoft",
      "paper_count": 81,
      "repo_count": 55
    },
    {
      "org": "openclaw",
      "paper_count": 74,
      "repo_count": 13
    },
    {
      "org": "openai",
      "paper_count": 73,
      "repo_count": 29
    },
    {
      "org": "huggingface",
      "paper_count": 66,
      "repo_count": 16
    },
    {
      "org": "yoheinakajima",
      "paper_count": 36,
      "repo_count": 3
    },
    {
      "org": "meta-llama",
      "paper_count": 35,
      "repo_count": 5
    },
    {
      "org": "google",
      "paper_count": 31,
      "repo_count": 27
    },
    {
      "org": "anthropics",
      "paper_count": 31,
      "repo_count": 13
    }
  ],
  "newest_papers": [
    {
      "object_id": "c11c5f29-2201-50b9-b225-9179fb173850",
      "arxiv_id": "2607.21557",
      "title": "OpenForgeRL: Train Harness-native Agents in Any Environment",
      "primary_category": "cs.AI",
      "published": "2026-07-23",
      "topic": "reinforcement learning",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "cd9db926-0547-5f47-9f24-77a311b1534c",
      "arxiv_id": "2607.21503",
      "title": "Agentic Context Management: Solving Agent Memory and Cost by Treating Them as Lifecycle and Architecture Problems",
      "primary_category": "cs.AI",
      "published": "2026-07-23",
      "topic": "reasoning",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "f28f11d3-1437-5317-becc-e6f9961059b7",
      "arxiv_id": "2607.21495",
      "title": "Toward Continuous Assurance for the Democratization of AI Agent Creation in Industry",
      "primary_category": "cs.AI",
      "published": "2026-07-23",
      "topic": "other",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "d627c8ef-af14-506a-aeb8-3a16b61d6fc4",
      "arxiv_id": "2607.21482",
      "title": "Agentic coding without the cloud: evaluating open-weight large language models on longitudinal data preparation tasks",
      "primary_category": "cs.AI",
      "published": "2026-07-23",
      "topic": "code generation",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "c6c328c4-8c2d-5e14-94d8-e4eb33732f62",
      "arxiv_id": "2607.21345",
      "title": "Regulating autonomous and agentic AI",
      "primary_category": "cs.AI",
      "published": "2026-07-23",
      "topic": "other",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Agentic AI"
    },
    {
      "object_id": "f5f47770-a66b-545e-a0d4-2234005435a9",
      "arxiv_id": "2607.21325",
      "title": "Toward cryptographically verifiable authorization for autonomous AI agents: A security hypothesis, preliminary formal model, and proof-of-concept implementation",
      "primary_category": "cs.CR",
      "published": "2026-07-23",
      "topic": "reasoning",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "550aa6de-fd66-5731-a3b6-dc85ab1ca1ef",
      "arxiv_id": "2607.21209",
      "title": "Explainability Framework for Policy-Aware Autonomous Agents",
      "primary_category": "cs.LO",
      "published": "2026-07-23",
      "topic": "planning",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "1478bc45-1ace-5f8f-8b15-538a935aea38",
      "arxiv_id": "2607.20926",
      "title": "SciExplore: Evaluating Autonomous Agents from Scientific Navigation to Information Integration",
      "primary_category": "cs.AI",
      "published": "2026-07-23",
      "topic": "reasoning",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "1d1ebe4b-e5c8-5111-b6df-12dae3674d59",
      "arxiv_id": "2607.20773",
      "title": "HARP: The Human--AI Research Platform",
      "primary_category": "cs.HC",
      "published": "2026-07-22",
      "topic": "other",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "3d87ece4-4ebf-5c91-a5c8-ea70fde9d0be",
      "arxiv_id": "2607.20709",
      "title": "NVIDIA-labs OO Agents: Native Python Object-Oriented Agents",
      "primary_category": "cs.AI",
      "published": "2026-07-22",
      "topic": "code generation",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "33dc38ad-7924-5a62-a9df-2773478d4140",
      "arxiv_id": "2607.20255",
      "title": "The Ethics of Autonomous AI Agents for Offensive Security",
      "primary_category": "cs.CR",
      "published": "2026-07-22",
      "topic": "other",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "d81b7aee-ea6e-5181-98c5-209ad93c8724",
      "arxiv_id": "2607.20582",
      "title": "Bayesian uncertainty estimation improves clinical decision making in medical AI agents",
      "primary_category": "cs.LG",
      "published": "2026-07-22",
      "topic": "image classification",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "c707a56a-8f92-5266-ac92-e9530a3dc9f7",
      "arxiv_id": "2607.19941",
      "title": "A Framework of User Experience Principles for Human-AI Agent Interaction in the Workplace",
      "primary_category": "cs.HC",
      "published": "2026-07-22",
      "topic": "other",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "aa36847b-fa5c-5934-8ecb-a4f785265fc9",
      "arxiv_id": "2607.19865",
      "title": "DocOps: A Verifiable Benchmark for Autonomous Agents in Complex Document Operations",
      "primary_category": "cs.AI",
      "published": "2026-07-22",
      "topic": "reasoning",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "2c8345cc-411d-5dd8-9f3f-575ef25932f5",
      "arxiv_id": "2607.19837",
      "title": "Know Your Agent: Reconnaissance-Driven Pentesting of AI Agents",
      "primary_category": "cs.AI",
      "published": "2026-07-22",
      "topic": "other",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "eeabd017-6e75-57c4-adee-c4285489ed54",
      "arxiv_id": "2607.19793",
      "title": "Silent Failures in Multimodal Agentic Search:A Diagnostic Taxonomy and Cross-Judge Evaluation",
      "primary_category": "cs.AI",
      "published": "2026-07-22",
      "topic": "reasoning",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "ReAct"
    },
    {
      "object_id": "9c081392-de8b-5f7a-ba6c-c0837e9b6e05",
      "arxiv_id": "2607.19747",
      "title": "Beyond Relevance-Centric Retrieval: Rubric-Oriented Document Set Selection and Ranking",
      "primary_category": "cs.CL",
      "published": "2026-07-22",
      "topic": "other",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "74a35ed6-95bf-59d9-bcc0-8e2bef860557",
      "arxiv_id": "2607.19321",
      "title": "ResearchArena: Evaluating Sabotage and Monitoring in Automated AI R&D",
      "primary_category": "cs.AI",
      "published": "2026-07-21",
      "topic": "other",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "e53a3f76-f6f0-5e85-af87-fd6067bcd3fc",
      "arxiv_id": "2607.19297",
      "title": "Graph-Based Agentic AI with LangGraph: Workflow Pathways for Long-Running Stateful Business Processes",
      "primary_category": "cs.AI",
      "published": "2026-07-21",
      "topic": "other",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Agentic AI"
    },
    {
      "object_id": "5e9416f8-5683-5720-bbe8-7b1c3e51268d",
      "arxiv_id": "2607.19262",
      "title": "BioSecBench-Surveillance: A Verifiable Benchmark for AI Agents in Pathogen Genomic Surveillance",
      "primary_category": "cs.AI",
      "published": "2026-07-21",
      "topic": "reasoning",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "203753d3-f1c3-5bf7-8bf3-dc86a9917b49",
      "arxiv_id": "2607.19453",
      "title": "Predictive Extrema, Unprofitable Policies: An AI-Assisted Audit of Candle-Based Binance Spot Timing Models",
      "primary_category": "cs.LG",
      "published": "2026-07-21",
      "topic": "other",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "973547cf-c483-5181-a43f-5e0a8cb2f5d7",
      "arxiv_id": "2607.18970",
      "title": "Skillware: A Software Ontology and Engineering Lifecycle for Persistent Behavioral Artifacts",
      "primary_category": "cs.SE",
      "published": "2026-07-21",
      "topic": "other",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "6fadfb1c-6c8e-50e7-9219-3001d488405b",
      "arxiv_id": "2607.18665",
      "title": "SciHazard: A Benchmark for Measuring Scientific Safety Risks with Decomposed Harm Scoring",
      "primary_category": "cs.AI",
      "published": "2026-07-21",
      "topic": "question answering",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "32339e60-4ce5-52f4-bece-f03f845a0f5a",
      "arxiv_id": "2607.19433",
      "title": "The Chronos Vulnerability: A Taxonomy of Temporal Persistence and Memory-Based Deception in Agentic AI",
      "primary_category": "cs.AI",
      "published": "2026-07-20",
      "topic": "reasoning",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Agentic AI"
    },
    {
      "object_id": "fdcaee74-0c3a-5e37-ba8d-2d5f4bda2389",
      "arxiv_id": "2607.19432",
      "title": "ChainWatch: A Kill Chain-Aligned Sequential Detection Framework for Multi-Step Attacks in MCP-Based AI Agent Systems",
      "primary_category": "cs.CR",
      "published": "2026-07-20",
      "topic": "other",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "5c901f13-b395-5c95-9b3d-04ea3349e799",
      "arxiv_id": "2607.18548",
      "title": "Engineering Trustworthy Agentic AI for Critical Systems",
      "primary_category": "cs.AI",
      "published": "2026-07-20",
      "topic": "survey",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Agentic AI"
    },
    {
      "object_id": "b620241f-6813-5ecc-b7fb-8a324d200eed",
      "arxiv_id": "2607.18366",
      "title": "Operational Hallucination and Safety Drift in AI Agents",
      "primary_category": "cs.AI",
      "published": "2026-07-20",
      "topic": "planning",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "ca5fa6b4-1dff-5610-a257-35c85fa9c420",
      "arxiv_id": "2607.18147",
      "title": "LLMs and Agentic AI Systems for Smart Grids: A Tutorial on Architectures and Applications",
      "primary_category": "eess.SY",
      "published": "2026-07-20",
      "topic": "reasoning",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Agentic AI"
    },
    {
      "object_id": "7e884189-073e-5d88-8a6c-c9e1f69681a4",
      "arxiv_id": "2607.18064",
      "title": "Autoresearch with Coding Agents: Generalizers and Metric-Maximizers on Quran Recitation Data",
      "primary_category": "cs.SE",
      "published": "2026-07-20",
      "topic": "other",
      "in_title": 0,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    },
    {
      "object_id": "7dc85494-5df8-535b-87a7-e9bf3490ec18",
      "arxiv_id": "2607.17986",
      "title": "Self-State Attacks on Self-Hosted AI Agents: How Far Can OS Defenses Go?",
      "primary_category": "cs.CR",
      "published": "2026-07-20",
      "topic": "other",
      "in_title": 1,
      "match_source": "entity",
      "match_entity": "Autonomous Agents"
    }
  ],
  "most_cited": [
    {
      "object_id": "7f5b7958-407e-50cb-a428-15dc8748f40a",
      "arxiv_id": "2210.03629",
      "title": "ReAct: Synergizing Reasoning and Acting in Language Models",
      "primary_category": "cs.CL",
      "published": "2022-10-06",
      "citation_count": 2391
    },
    {
      "object_id": "2a10a706-e7cb-5a89-a8e7-199410496641",
      "arxiv_id": "2307.13854",
      "title": "WebArena: A Realistic Web Environment for Building Autonomous Agents",
      "primary_category": "cs.AI",
      "published": "2023-07-25",
      "citation_count": 873
    },
    {
      "object_id": "0c83402a-6d82-57aa-a44e-afe6bec3af9c",
      "arxiv_id": "2307.16789",
      "title": "ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs",
      "primary_category": "cs.AI",
      "published": "2023-07-31",
      "citation_count": 842
    },
    {
      "object_id": "f1740b36-9b18-5bea-a295-ff96e4df18ca",
      "arxiv_id": "2310.02255",
      "title": "MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts",
      "primary_category": "cs.CV",
      "published": "2023-10-03",
      "citation_count": 791
    },
    {
      "object_id": "6e090b12-1c7c-583d-a1b5-4b634b864121",
      "arxiv_id": "2308.03688",
      "title": "AgentBench: Evaluating LLMs as Agents",
      "primary_category": "cs.AI",
      "published": "2023-08-07",
      "citation_count": 752
    },
    {
      "object_id": "71a1c990-2245-50e1-a21e-c88fc775ad1c",
      "arxiv_id": "2408.06292",
      "title": "The AI Scientist: Towards Fully Automated Open-Ended Scientific Discovery",
      "primary_category": "cs.AI",
      "published": "2024-08-12",
      "citation_count": 714
    },
    {
      "object_id": "3c83ce01-e8be-5a01-9768-1828937c8dd4",
      "arxiv_id": "2504.19413",
      "title": "Mem0: Building Production-Ready AI Agents with Scalable Long-Term Memory",
      "primary_category": "cs.CL",
      "published": "2025-04-28",
      "citation_count": 639
    },
    {
      "object_id": "086eb207-09b5-5041-9e70-ad7c45562036",
      "arxiv_id": "2406.12045",
      "title": "$τ$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains",
      "primary_category": "cs.AI",
      "published": "2024-06-17",
      "citation_count": 590
    }
  ],
  "links": {
    "page": "https://brunosan.de/arxiv/agents/",
    "json": "https://brunosan.de/arxiv/tracks/ai-agents.json",
    "mcp": "https://arxiv.mcp.brunosan.de/mcp"
  }
}