{
  "id": "webarena",
  "type": "entity",
  "name": "WebArena",
  "summary": "A self-hosted, realistic web environment for evaluating autonomous agents: functional sites for e-commerce, forum discussion, collaborative software development and content management, plus maps and knowledge-base tools. Tasks are long-horizon and graded on functional correctness of the end state. Headline result at publication: best GPT-4 agent 14.41% against human 78.24%.",
  "locale": "en",
  "tags": [
    "webarena",
    "benchmarks",
    "web-agents",
    "evaluation",
    "sandbox"
  ],
  "relations": [
    {
      "rel": "related",
      "target": "osworld"
    },
    {
      "rel": "related",
      "target": "sandboxed-execution"
    },
    {
      "rel": "related",
      "target": "measuring-agent-reliability"
    }
  ],
  "questions": [
    "What is WebArena?",
    "How does WebArena grade agent success?"
  ],
  "claims": [
    {
      "id": "c1",
      "text": "WebArena provides realistic self-hosted environments across four domains — e-commerce, social forums, collaborative software development and content management — with tool and knowledge-base access, so agent evaluation runs against functional sites rather than static snapshots.",
      "sources": [
        "webarena-paper"
      ],
      "confidence": 0.95
    },
    {
      "id": "c2",
      "text": "WebArena grades functional correctness of task completion on long-horizon tasks; at publication the best GPT-4-based agent reached 14.41% end-to-end success against 78.24% for humans.",
      "sources": [
        "webarena-paper"
      ],
      "confidence": 0.9
    }
  ],
  "takeaways": [],
  "faqs": [],
  "evidence_tier": "secondary",
  "evidence": {
    "level": "benchmark",
    "source_types": [
      "benchmark",
      "paper"
    ]
  },
  "moat_flag": false,
  "winning_edge": "Frames WebArena as the reference implementation of two principles this corpus argues for — contained realistic environments and end-state grading — with the numbers to justify both.",
  "confidence": 0.85,
  "last_verified": "2026-08-08",
  "canonical_url": "https://agentreliability.dev/k/webarena",
  "api_url": "https://agentreliability.dev/api/k/webarena.json",
  "jsonld": {
    "@context": "https://schema.org",
    "name": "WebArena",
    "description": "A self-hosted, realistic web environment for evaluating autonomous agents: functional sites for e-commerce, forum discussion, collaborative software development and content management, plus maps and knowledge-base tools. Tasks are long-horizon and graded on functional correctness of the end state. Headline result at publication: best GPT-4 agent 14.41% against human 78.24%.",
    "url": "https://agentreliability.dev/k/webarena",
    "license": "https://spdx.org/licenses/CC-BY-4.0.html",
    "dateModified": "2026-08-08",
    "citation": [
      {
        "@type": "CreativeWork",
        "name": "WebArena: A Realistic Web Environment for Building Autonomous Agents",
        "url": "https://arxiv.org/abs/2307.13854"
      }
    ],
    "author": {
      "@type": "Person",
      "name": "Santiago Santa María Morales",
      "jobTitle": "practitioner — harness engineering and agent evaluation in production"
    },
    "@type": "Article",
    "headline": "WebArena"
  }
}
