{
  "admissionBoundary": "No current score, rank or model-performance claim is admitted until an immutable observation passes schema validation, exact entity resolution, source-receipt verification, rights review and the comparison gate.",
  "apifyReceipt": "config/benchmarks/apify-research-wave84.json",
  "benchmarkProfiles": [
    {
      "benchmarkEntityId": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "comparisonBoundary": "Different suites, scenarios, adapters, prompt formats, instance caps, perturbation policies or metrics are CONTEXT_ONLY.",
      "id": "helm",
      "name": "HELM",
      "passport": "/entities/ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589/",
      "prohibitedClaims": [
        "a bare HELM score",
        "contamination-free because the registry is silent"
      ],
      "requiredIdentity": [
        "framework release or commit",
        "suite",
        "scenario",
        "run entry",
        "model adapter"
      ],
      "requiredProtocol": [
        "RunSpec",
        "ScenarioSpec",
        "AdapterSpec",
        "ExecutionSpec",
        "MetricSpec",
        "instance cap",
        "perturbation and seed policy"
      ],
      "sourceReferences": [
        {
          "evidenceRole": "identity and protocol",
          "locator": "RunSpec, ScenarioSpec, AdapterSpec, ExecutionSpec and MetricSpec structure",
          "url": "https://crfm-helm.readthedocs.io/en/latest/code/"
        },
        {
          "evidenceRole": "integrity boundary",
          "locator": "contamination registry and its bounded evidence role",
          "url": "https://github.com/stanford-crfm/helm/blob/main/src/helm/benchmark/static/contamination.yaml"
        }
      ],
      "version": "source revision 63754d05; no semantic benchmark version asserted"
    },
    {
      "benchmarkEntityId": "ENT-a5bdb209-876d-458a-bd75-4b89e7dfedd6",
      "comparisonBoundary": "Round, workload, division, scenario, accuracy tier, availability and power method require an explicit compatibility statement.",
      "id": "mlperf",
      "name": "MLPerf",
      "passport": "/entities/ENT-a5bdb209-876d-458a-bd75-4b89e7dfedd6/",
      "prohibitedClaims": [
        "unverified result described as official",
        "cross-scenario fastest-system claim"
      ],
      "requiredIdentity": [
        "benchmark family and round",
        "rules commit",
        "workload",
        "division",
        "system category",
        "scenario",
        "submission ID"
      ],
      "requiredProtocol": [
        "LoadGen revision",
        "accuracy tier",
        "availability category",
        "power method",
        "system description",
        "official change state"
      ],
      "sourceReferences": [
        {
          "evidenceRole": "identity and protocol",
          "locator": "divisions, scenarios, LoadGen, accuracy and submission rules",
          "url": "https://github.com/mlcommons/inference_policies/blob/master/inference_rules.adoc"
        },
        {
          "evidenceRole": "comparison boundary",
          "locator": "cross-version result compatibility rules",
          "url": "https://github.com/mlcommons/policies/blob/master/MLPerf_Compatibility_Table.adoc"
        },
        {
          "evidenceRole": "prohibited claims",
          "locator": "public result wording and verification constraints",
          "url": "https://github.com/mlcommons/policies/blob/master/MLPerf_Results_Messaging_Guidelines.adoc"
        }
      ],
      "version": "6.0"
    },
    {
      "benchmarkEntityId": "ENT-b2bceffe-2ce0-475c-b161-529c13553b46",
      "comparisonBoundary": "Full, Lite, Verified, Multilingual and Multimodal are distinct; best@k, pass@k and pass@1 are not interchangeable.",
      "id": "swe-bench",
      "name": "SWE-bench",
      "passport": "/entities/ENT-b2bceffe-2ce0-475c-b161-529c13553b46/",
      "prohibitedClaims": [
        "hidden-test access",
        "gold patch or solution browsing",
        "passed tests proves production correctness"
      ],
      "requiredIdentity": [
        "dataset variant and revision",
        "instance IDs",
        "repository base commits",
        "harness commit",
        "container digests"
      ],
      "requiredProtocol": [
        "scaffold commit",
        "model snapshot",
        "prompt",
        "tools and network",
        "token and time budget",
        "attempt and selection policy",
        "patch parser"
      ],
      "sourceReferences": [
        {
          "evidenceRole": "identity and protocol",
          "locator": "evaluation harness, predictions, containers and grading",
          "url": "https://github.com/SWE-bench/SWE-bench/blob/main/docs/guides/evaluation.md"
        },
        {
          "evidenceRole": "integrity and comparison boundary",
          "locator": "submission attempts, anti-solution-browsing and disclosure checklist",
          "url": "https://github.com/swe-bench/experiments/blob/main/checklist.md"
        }
      ],
      "version": "Verified subset; source revision 02e7a74f"
    },
    {
      "benchmarkEntityId": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "comparisonBoundary": "pass@k requires n, c, k and estimator details; pass@k is never silently restated as pass@1.",
      "id": "humaneval",
      "name": "HumanEval",
      "passport": "/entities/ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197/",
      "prohibitedClaims": [
        "general software-engineering ability",
        "safe execution",
        "contamination-free"
      ],
      "requiredIdentity": [
        "dataset or repository commit",
        "task bytes",
        "Python environment",
        "evaluator commit"
      ],
      "requiredProtocol": [
        "samples per task",
        "temperature and decoding",
        "k",
        "prompt format",
        "parser",
        "timeout and sandbox"
      ],
      "sourceReferences": [
        {
          "evidenceRole": "metric contract",
          "locator": "pass@k estimator and sample accounting",
          "url": "https://github.com/openai/human-eval/blob/master/human_eval/evaluation.py"
        },
        {
          "evidenceRole": "prohibited claims",
          "locator": "execution warning that reliability_guard is not a security sandbox",
          "url": "https://github.com/openai/human-eval/blob/master/human_eval/execution.py"
        }
      ],
      "version": "source revision 6d43fb98; no semantic benchmark version stated"
    },
    {
      "benchmarkEntityId": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "comparisonBoundary": "Direct comparison requires identical data bytes, subjects, shots, scoring path and aggregation.",
      "id": "mmlu",
      "name": "MMLU",
      "passport": "/entities/ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70/",
      "prohibitedClaims": [
        "general intelligence",
        "expert-level capability without subject qualification",
        "contamination-free"
      ],
      "requiredIdentity": [
        "repository commit",
        "data digest",
        "exact subjects",
        "test split"
      ],
      "requiredProtocol": [
        "shots actually used",
        "demonstration IDs and order",
        "prompt and chat wrapper",
        "answer-choice order",
        "probability or generation scoring",
        "macro or micro aggregation"
      ],
      "sourceReferences": [
        {
          "evidenceRole": "identity and protocol",
          "locator": "subject files, few-shot formatting and answer-token scoring",
          "url": "https://github.com/hendrycks/test/blob/master/evaluate.py"
        },
        {
          "evidenceRole": "comparison boundary",
          "locator": "material differences among MMLU prompting and scoring protocols",
          "url": "https://crfm.stanford.edu/2024/05/01/helm-mmlu.html"
        }
      ],
      "version": "source revision 4450500f; no semantic benchmark version stated"
    },
    {
      "benchmarkEntityId": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "comparisonBoundary": "Main and Diamond, or retrieval and closed-book protocols, are CONTEXT_ONLY rather than silently pooled.",
      "id": "gpqa",
      "name": "GPQA",
      "passport": "/entities/ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a/",
      "prohibitedClaims": [
        "republishing gated question text",
        "Google-proof as a timeless guarantee"
      ],
      "requiredIdentity": [
        "repository and data revision",
        "Main, Extended or Diamond slice",
        "item digest",
        "correction state",
        "answer-order seed"
      ],
      "requiredProtocol": [
        "prompt mode",
        "examples and chain-of-thought policy",
        "open or closed book",
        "retrieval and tools",
        "parser and abstention policy"
      ],
      "sourceReferences": [
        {
          "evidenceRole": "identity and metric",
          "locator": "Main, Extended and Diamond dataset organization",
          "url": "https://github.com/idavidrein/gpqa"
        },
        {
          "evidenceRole": "integrity and publication boundary",
          "locator": "gated access, canary and non-redistribution conditions",
          "url": "https://huggingface.co/datasets/Idavidrein/gpqa"
        }
      ],
      "version": "source revision 56686c06; no semantic benchmark version stated"
    },
    {
      "benchmarkEntityId": "ENT-3a6be454-fedf-4237-8d71-f378e916b0a7",
      "comparisonBoundary": "Ratings from independently anchored snapshots or different categories and methods are not placed on one absolute scale.",
      "id": "arena",
      "name": "Arena",
      "passport": "/entities/ENT-3a6be454-fedf-4237-8d71-f378e916b0a7/",
      "prohibitedClaims": [
        "rank without interval",
        "universal quality",
        "safety, factuality, latency or cost inferred from preference"
      ],
      "requiredIdentity": [
        "product and category",
        "vote cutoff interval",
        "battle snapshot digest",
        "model aliases and endpoints",
        "pipeline and policy revisions"
      ],
      "requiredProtocol": [
        "eligibility and filtering",
        "sampling probabilities",
        "style control",
        "rating implementation",
        "bootstrap or uncertainty method",
        "battle graph"
      ],
      "sourceReferences": [
        {
          "evidenceRole": "identity and publication state",
          "locator": "live evaluation, vote thresholds, preliminary results and policy revisions",
          "url": "https://arena.ai/blog/policy"
        },
        {
          "evidenceRole": "metric and comparison boundary",
          "locator": "pairwise evaluation, Bradley-Terry estimation, bootstrap uncertainty and limitations",
          "url": "https://arxiv.org/html/2403.04132"
        }
      ],
      "version": "last updated 2026-09-01"
    },
    {
      "benchmarkEntityId": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "comparisonBoundary": "Suite, task labels, infrastructure, scaffold, elicitation, budgets, reliability target and analysis model must align.",
      "id": "metr-time-horizon",
      "name": "METR Time Horizon",
      "passport": "/entities/ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e/",
      "prohibitedClaims": [
        "maximum autonomous duration",
        "automation percentage",
        "deployment timeline forecast"
      ],
      "requiredIdentity": [
        "suite and revision",
        "task IDs",
        "human-time baseline version",
        "evaluation infrastructure",
        "analysis code and model version"
      ],
      "requiredProtocol": [
        "scaffold commit",
        "tools and network",
        "elicitation procedure",
        "token and time budget",
        "run count",
        "reliability target"
      ],
      "sourceReferences": [
        {
          "evidenceRole": "metric contract",
          "locator": "time-horizon definition, reliability target and modelling procedure",
          "url": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/"
        },
        {
          "evidenceRole": "protocol and integrity",
          "locator": "scaffold, tools, elicitation, red flags and validity reporting",
          "url": "https://metr.org/blog/2024-03-15-guidelines-for-capability-elicitation/"
        },
        {
          "evidenceRole": "prohibited claims",
          "locator": "limits on autonomy, automation and forecasting interpretations",
          "url": "https://metr.org/notes/2026-01-22-time-horizon-limitations/"
        }
      ],
      "version": "1.1"
    }
  ],
  "hardEqualityDimensions": [
    "observation kind and result-status class",
    "evaluated model/deployment identity and full system snapshot",
    "benchmark identity, revision and variant",
    "metric definition, unit, direction, aggregation and parameters",
    "evaluation harness and dataset revision",
    "prompting, decoding, scaffold, tools, budget and attempt policy",
    "sample population and denominator",
    "data cutoff",
    "integrity treatment and contamination status",
    "verification status and scope"
  ],
  "levels": {
    "BLOCKED": "Identity, rights, integrity, verification or validity defects prohibit comparison publication.",
    "CONDITIONED": "A disclosed, source-supported transformation permits a qualified comparison; assumptions and residual differences remain visible.",
    "CONTEXT_ONLY": "The records may be juxtaposed as separate facts, but no quantitative superiority or trend inference is permitted.",
    "DIRECT": "Both observations pass the admission contract and every comparison-key dimension is exactly aligned."
  },
  "normalizationBoundary": "A normalization never erases protocol differences. It must carry a source receipt, assumptions and residual limitations.",
  "numericObservationsAdmitted": 0,
  "rankingPolicy": "No rank is stored in the observation schema. Any future ordering must be derived from a named, reproducible view over comparison-compatible observations.",
  "researchCutoff": "2026-09-03",
  "researchDocument": "docs/research/AI_BENCHMARK_RESULT_OBSERVATION_WAVE84.md",
  "schema": "metatron.intelligence.benchmark-comparison-policy.v1",
  "version": "1.0.0"
}
