{
  "id": "independent-evaluation",
  "type": "entity",
  "name": "Independent third-party evaluation",
  "summary": "Evaluation performed by organizations that did not build the system: external red teams, research nonprofits running autonomy evaluations, and government institutes publishing open harnesses. Self-evaluation has a structural conflict of interest — the same incentive gradient Goodhart describes — and the emerging ecosystem of independent evaluators is the field's answer. For buyers, a vendor's willingness to be independently evaluated is itself a reliability signal.",
  "locale": "en",
  "tags": [
    "governance",
    "third-party",
    "auditing",
    "evaluation",
    "ecosystem"
  ],
  "relations": [
    {
      "rel": "related",
      "target": "goodhart-resistance"
    },
    {
      "rel": "related",
      "target": "red-teaming-your-agent"
    },
    {
      "rel": "related",
      "target": "inspect-eval-framework"
    }
  ],
  "questions": [
    "Who evaluates AI systems besides their builders?",
    "Why does independent evaluation matter for agents?",
    "What should I ask a vendor about third-party testing?"
  ],
  "claims": [
    {
      "id": "c1",
      "text": "Frontier-lab practice already includes external capacity: Anthropic's red-teaming account recommends funding standards development, supporting independent testing organizations, professionalizing red teaming with certification, and giving vetted third parties access to systems.",
      "sources": [
        "anthropic-red-teaming"
      ],
      "confidence": 0.9
    },
    {
      "id": "c2",
      "text": "Independent evaluators exist and publish: METR runs autonomy evaluations of frontier models from multiple labs, partnering with developers while also conducting independent assessments of publicly released models.",
      "sources": [
        "metr"
      ],
      "confidence": 0.85
    },
    {
      "id": "c3",
      "text": "Public evaluation infrastructure lowers the barrier: the UK AI Security Institute maintains Inspect as open source with 200+ prebuilt evaluations, so third-party evaluation does not require third-party tooling from scratch.",
      "sources": [
        "inspect-ai"
      ],
      "confidence": 0.85
    }
  ],
  "takeaways": [],
  "faqs": [],
  "evidence_tier": "secondary",
  "evidence": {
    "level": "industry_observation",
    "source_types": [
      "industry_observation"
    ]
  },
  "moat_flag": false,
  "winning_edge": "Frames independence as a Goodhart countermeasure and gives buyers a concrete due-diligence question set, where existing coverage describes the ecosystem without saying what to do with it.",
  "confidence": 0.8,
  "last_verified": "2026-08-08",
  "canonical_url": "https://agentreliability.dev/k/independent-evaluation",
  "api_url": "https://agentreliability.dev/api/k/independent-evaluation.json",
  "jsonld": {
    "@context": "https://schema.org",
    "name": "Independent third-party evaluation",
    "description": "Evaluation performed by organizations that did not build the system: external red teams, research nonprofits running autonomy evaluations, and government institutes publishing open harnesses. Self-evaluation has a structural conflict of interest — the same incentive gradient Goodhart describes — and the emerging ecosystem of independent evaluators is the field's answer. For buyers, a vendor's willingness to be independently evaluated is itself a reliability signal.",
    "url": "https://agentreliability.dev/k/independent-evaluation",
    "license": "https://spdx.org/licenses/CC-BY-4.0.html",
    "dateModified": "2026-08-08",
    "citation": [
      {
        "@type": "CreativeWork",
        "name": "Inspect — evaluation framework for large language models",
        "url": "https://inspect.aisi.org.uk/"
      },
      {
        "@type": "CreativeWork",
        "name": "Challenges in red teaming AI systems",
        "url": "https://www.anthropic.com/news/challenges-in-red-teaming-ai-systems"
      },
      {
        "@type": "CreativeWork",
        "name": "METR — Model Evaluation & Threat Research",
        "url": "https://metr.org/"
      }
    ],
    "author": {
      "@type": "Person",
      "name": "Santiago Santa María Morales",
      "jobTitle": "practitioner — harness engineering and agent evaluation in production"
    },
    "@type": "Article",
    "headline": "Independent third-party evaluation"
  }
}
