{
  "id": "osworld",
  "type": "entity",
  "name": "OSWorld",
  "summary": "A benchmark of 369 open-ended tasks executed in real operating systems (Ubuntu, Windows, macOS) spanning web and desktop apps, file I/O and multi-application workflows. Every task ships its own initial-state setup and an execution-based evaluation script, making it a working template for reproducible computer-use agent evaluation. Headline gap at publication: humans 72.36%, best model 12.24%.",
  "locale": "en",
  "tags": [
    "osworld",
    "benchmarks",
    "computer-use",
    "evaluation",
    "execution"
  ],
  "relations": [
    {
      "rel": "related",
      "target": "webarena"
    },
    {
      "rel": "related",
      "target": "sandboxed-execution"
    },
    {
      "rel": "related",
      "target": "measuring-agent-reliability"
    }
  ],
  "questions": [
    "What is OSWorld?",
    "How are computer-use agents benchmarked?"
  ],
  "claims": [
    {
      "id": "c1",
      "text": "OSWorld comprises 369 real computer tasks — web and desktop apps in open domains, OS file I/O, and workflows spanning multiple applications — running on real operating systems including Ubuntu, Windows and macOS.",
      "sources": [
        "osworld-paper"
      ],
      "confidence": 0.95
    },
    {
      "id": "c2",
      "text": "Each OSWorld task defines a detailed initial-state setup plus a custom execution-based evaluation script, so grading is reproducible and independent of the agent's narration.",
      "sources": [
        "osworld-paper"
      ],
      "confidence": 0.9
    },
    {
      "id": "c3",
      "text": "At publication humans accomplished over 72.36% of OSWorld tasks against 12.24% for the best model, with difficulties concentrated in GUI grounding and operational knowledge.",
      "sources": [
        "osworld-paper"
      ],
      "confidence": 0.9
    }
  ],
  "takeaways": [],
  "faqs": [],
  "evidence_tier": "secondary",
  "evidence": {
    "level": "benchmark",
    "source_types": [
      "benchmark",
      "paper"
    ]
  },
  "moat_flag": false,
  "winning_edge": "Reads OSWorld as an evaluation-design template — seeded initial state plus execution-based grader per task — rather than as another leaderboard, and names where computer-use agents actually fail.",
  "confidence": 0.85,
  "last_verified": "2026-08-08",
  "canonical_url": "https://agentreliability.dev/k/osworld",
  "api_url": "https://agentreliability.dev/api/k/osworld.json",
  "jsonld": {
    "@context": "https://schema.org",
    "name": "OSWorld",
    "description": "A benchmark of 369 open-ended tasks executed in real operating systems (Ubuntu, Windows, macOS) spanning web and desktop apps, file I/O and multi-application workflows. Every task ships its own initial-state setup and an execution-based evaluation script, making it a working template for reproducible computer-use agent evaluation. Headline gap at publication: humans 72.36%, best model 12.24%.",
    "url": "https://agentreliability.dev/k/osworld",
    "license": "https://spdx.org/licenses/CC-BY-4.0.html",
    "dateModified": "2026-08-08",
    "citation": [
      {
        "@type": "CreativeWork",
        "name": "OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments",
        "url": "https://arxiv.org/abs/2404.07972"
      }
    ],
    "author": {
      "@type": "Person",
      "name": "Santiago Santa María Morales",
      "jobTitle": "practitioner — harness engineering and agent evaluation in production"
    },
    "@type": "Article",
    "headline": "OSWorld"
  }
}
