{
  "schema": "the-split.event-dossier.v0",
  "generatedAt": "2026-07-26T10:11:38.627Z",
  "event": {
    "slug": "model-evaluation-transparency",
    "title": "Open evaluation tools turn model transparency into shared test infrastructure",
    "category": "Tech / AI",
    "region": "United Kingdom / United States / Global AI labs",
    "location": "London / Gaithersburg / global evaluation community",
    "coordinates": {
      "lat": 51.5074,
      "lng": -0.1278
    },
    "status": "draft",
    "sensitivity": "high",
    "updatedAt": "2024-11-13T00:00:00Z"
  },
  "summaryForAgents": "Open evaluation tools turn model transparency into shared test infrastructure. Category: Tech / AI. Region: United Kingdom / United States / Global AI labs. The dossier separates shared facts from Atlantic/Eurasian framing and Bridge uncertainty. Uses named source entries; still verify context and URLs before publication.",
  "sharedFacts": [
    "GOV.UK says the UK AI Safety Institute released Inspect on 10 May 2024 as an open-source evaluations platform intended to strengthen and accelerate global AI safety evaluations.",
    "AISI says Inspect Evals, announced on 13 November 2024, made dozens of community-contributed LLM evaluations available, covering domains such as coding, mathematics, cybersecurity, safeguards, reasoning and general knowledge.",
    "NIST describes ARIA as an evaluation environment for assessing risks and impacts of AI across model testing, red-teaming and field testing, moving beyond performance and accuracy toward technical and contextual robustness.",
    "NIST describes Dioptra as a software test platform for assessing trustworthy AI characteristics and supporting the Measure function of the NIST AI Risk Management Framework through experiment design, execution and tracking."
  ],
  "lenses": {
    "atlantic": {
      "title": "Atlantic Lens",
      "summary": "Frames open evaluation tooling as public infrastructure for safety testing, procurement and frontier-model oversight.",
      "framing": [
        "open evaluations",
        "testing infrastructure",
        "procurement"
      ],
      "body": "Atlantic governance framing can treat Inspect, ARIA and Dioptra as practical evaluation plumbing: reusable tasks, test environments, logs, red-team workflows and experiment tracking that help regulators, labs and enterprise buyers ask more comparable safety questions. The source record supports evaluation infrastructure and open tooling; it does not prove that benchmark results fully predict real-world safety."
    },
    "eurasian": {
      "title": "Eurasian Lens",
      "summary": "Frames shared evaluation stacks as standards power that may spread capability while shaping whose tests count.",
      "framing": [
        "standards power",
        "open-source access",
        "sovereignty"
      ],
      "body": "Eurasian and Global South framing can see open evaluation repositories as a useful way to lower the barrier to model testing, while also asking whether UK- and U.S.-anchored toolchains define the evaluation agenda for everyone else. This remains interpretation: the named sources emphasize collaboration and community use, not a settled global governance mandate."
    },
    "bridge": {
      "title": "Bridge",
      "summary": "The verified core is open evaluation infrastructure; validity, coverage and policy consequences remain open.",
      "framing": [
        "measurement tooling",
        "coverage uncertain",
        "not certification"
      ],
      "body": "Both lenses can agree that model oversight is becoming more concrete when test suites, sandboxes and evaluation environments are public enough for reuse and critique. The cautious line is that these tools create evidence hooks for agents, auditors and governments, while the hard questions remain benchmark validity, gaming, hidden deployment context, model-provider documentation quality and whether evaluation findings lead to actual release or mitigation decisions."
    }
  },
  "claims": [
    {
      "id": "model-evaluation-transparency-fact-1",
      "type": "fact",
      "lens": "Bridge",
      "text": "GOV.UK says the UK AI Safety Institute released Inspect on 10 May 2024 as an open-source evaluations platform intended to strengthen and accelerate global AI safety evaluations.",
      "confidence": "medium",
      "sourceRefs": [
        "GOV.UK: AI Safety Institute releases Inspect evaluations platform",
        "AISI: Announcing Inspect Evals",
        "NIST ARIA: Assessing Risks and Impacts of AI",
        "NIST data publication: Dioptra Test Platform"
      ]
    },
    {
      "id": "model-evaluation-transparency-fact-2",
      "type": "fact",
      "lens": "Bridge",
      "text": "AISI says Inspect Evals, announced on 13 November 2024, made dozens of community-contributed LLM evaluations available, covering domains such as coding, mathematics, cybersecurity, safeguards, reasoning and general knowledge.",
      "confidence": "medium",
      "sourceRefs": [
        "GOV.UK: AI Safety Institute releases Inspect evaluations platform",
        "AISI: Announcing Inspect Evals",
        "NIST ARIA: Assessing Risks and Impacts of AI",
        "NIST data publication: Dioptra Test Platform"
      ]
    },
    {
      "id": "model-evaluation-transparency-fact-3",
      "type": "fact",
      "lens": "Bridge",
      "text": "NIST describes ARIA as an evaluation environment for assessing risks and impacts of AI across model testing, red-teaming and field testing, moving beyond performance and accuracy toward technical and contextual robustness.",
      "confidence": "medium",
      "sourceRefs": [
        "GOV.UK: AI Safety Institute releases Inspect evaluations platform",
        "AISI: Announcing Inspect Evals",
        "NIST ARIA: Assessing Risks and Impacts of AI",
        "NIST data publication: Dioptra Test Platform"
      ]
    },
    {
      "id": "model-evaluation-transparency-fact-4",
      "type": "fact",
      "lens": "Bridge",
      "text": "NIST describes Dioptra as a software test platform for assessing trustworthy AI characteristics and supporting the Measure function of the NIST AI Risk Management Framework through experiment design, execution and tracking.",
      "confidence": "medium",
      "sourceRefs": [
        "GOV.UK: AI Safety Institute releases Inspect evaluations platform",
        "AISI: Announcing Inspect Evals",
        "NIST ARIA: Assessing Risks and Impacts of AI",
        "NIST data publication: Dioptra Test Platform"
      ]
    },
    {
      "id": "model-evaluation-transparency-atlantic-frame",
      "type": "interpretation",
      "lens": "Atlantic",
      "text": "Frames open evaluation tooling as public infrastructure for safety testing, procurement and frontier-model oversight.",
      "confidence": "medium",
      "sourceRefs": [
        "NIST ARIA: Assessing Risks and Impacts of AI",
        "NIST data publication: Dioptra Test Platform"
      ]
    },
    {
      "id": "model-evaluation-transparency-eurasian-frame",
      "type": "interpretation",
      "lens": "Eurasian",
      "text": "Frames shared evaluation stacks as standards power that may spread capability while shaping whose tests count.",
      "confidence": "medium",
      "sourceRefs": [
        "AISI: Announcing Inspect Evals"
      ]
    },
    {
      "id": "model-evaluation-transparency-bridge-gap",
      "type": "unknown",
      "lens": "Bridge",
      "text": "The verified core is open evaluation infrastructure; validity, coverage and policy consequences remain open.",
      "confidence": "low",
      "sourceRefs": [
        "GOV.UK: AI Safety Institute releases Inspect evaluations platform",
        "AISI: Announcing Inspect Evals",
        "NIST ARIA: Assessing Risks and Impacts of AI",
        "NIST data publication: Dioptra Test Platform"
      ]
    }
  ],
  "timeline": [
    {
      "at": "2024-11-13T00:00:00Z",
      "title": "Event dossier updated",
      "note": "Dossier seeded from named official sources. Continue verifying context before publication.",
      "confidence": "high"
    },
    {
      "at": "2024-11-13T00:00:00Z",
      "title": "Lens comparison generated",
      "note": "Atlantic tags: open evaluations, testing infrastructure, procurement. Eurasian tags: standards power, open-source access, sovereignty.",
      "confidence": "medium"
    }
  ],
  "sources": [
    {
      "name": "GOV.UK: AI Safety Institute releases Inspect evaluations platform",
      "lens": "Bridge",
      "type": "official",
      "url": "https://www.gov.uk/government/news/ai-safety-institute-releases-new-ai-safety-evaluations-platform",
      "publishedAt": "2024-05-10",
      "note": "UK government press release announcing the open-source Inspect testing platform; says Inspect helps groups develop evaluations, assess model capabilities and produce scores across knowledge, reasoning and autonomous capabilities.",
      "lastCheckedAt": "2026-07-26",
      "lastCheckStatus": "reachable",
      "lastCheckNote": "Python urllib HEAD with dossier-audit user agent returned HTTP 200; web_extract retrieved the publication date, open-source release, global evaluation-collaboration framing and Inspect capability description."
    },
    {
      "name": "AISI: Announcing Inspect Evals",
      "lens": "Eurasian",
      "type": "official",
      "url": "https://www.aisi.gov.uk/blog/inspect-evals",
      "publishedAt": "2024-11-13",
      "note": "AI Security Institute post announcing a repository of community-contributed LLM benchmark evaluations spanning coding, mathematics, cybersecurity, safeguards, reasoning, general knowledge and common frontier-provider benchmarks.",
      "lastCheckedAt": "2026-07-26",
      "lastCheckStatus": "reachable",
      "lastCheckNote": "Python urllib HEAD with dossier-audit user agent returned HTTP 200; web_extract retrieved the Inspect Evals launch date, domain coverage, community-contribution purpose and note that Inspect AI was open sourced in May 2024."
    },
    {
      "name": "NIST ARIA: Assessing Risks and Impacts of AI",
      "lens": "Atlantic",
      "type": "official",
      "url": "https://ai-challenges.nist.gov/aria",
      "note": "Official NIST AI Challenges page for ARIA; describes model testing, red-teaming and field testing, technical/contextual robustness, sector-agnostic evaluation and an LLM-focused pilot schedule.",
      "lastCheckedAt": "2026-07-26",
      "lastCheckStatus": "reachable",
      "lastCheckNote": "Python urllib HEAD with dossier-audit user agent returned HTTP 200; web_extract retrieved the ARIA overview, three testing levels, robustness framing and pilot schedule."
    },
    {
      "name": "NIST data publication: Dioptra Test Platform",
      "lens": "Atlantic",
      "type": "official",
      "url": "https://www.nist.gov/data-publications/dioptra-test-platform",
      "publishedAt": "2024-07-24",
      "note": "NIST data-publication record for Dioptra; describes source code, technical documentation and examples for a test platform assessing trustworthy AI characteristics and tracking AI-risk experiments.",
      "lastCheckedAt": "2026-07-26",
      "lastCheckStatus": "reachable",
      "lastCheckNote": "Python urllib HEAD with dossier-audit user agent returned HTTP 200; Python GET retrieved NIST JSON metadata including DOI 10.18434/mds2-3398, title, modified date, trustworthy-AI description and REST/API experiment-tracking use cases."
    }
  ],
  "sourceAudit": {
    "sourcePolicy": "source-backed",
    "namedSourceCount": 4,
    "prototypeSourceCount": 0,
    "urlSourceCount": 4,
    "latestSourcePublishedAt": "2024-11-13",
    "oldestSourcePublishedAt": "2024-05-10",
    "missingPublishedAtCount": 1,
    "checkedSourceCount": 4,
    "reachableSourceCount": 4,
    "blockedSourceCount": 0,
    "latestSourceCheckedAt": "2026-07-26",
    "recommendedRecheckAfter": "2025-02-11",
    "freshnessNote": "All source entries are named and URL-backed; re-check links, dates and surrounding context before publication or reuse."
  },
  "caveat": "Source-backed seed dossier. Verify source URLs and surrounding context before publication."
}
