{
  "schema": "metatron.intelligence.benchmark-comparison-policy.v1",
  "version": "1.0.0",
  "researchCutoff": "2026-09-03",
  "researchDocument": "docs/research/AI_BENCHMARK_RESULT_OBSERVATION_WAVE84.md",
  "numericObservationsAdmitted": 0,
  "levels": {
    "DIRECT": "Both observations pass the admission contract and every comparison-key dimension is exactly aligned.",
    "CONDITIONED": "A disclosed, source-supported transformation permits a qualified comparison; assumptions and residual differences remain visible.",
    "CONTEXT_ONLY": "The records may be juxtaposed as separate facts, but no quantitative superiority or trend inference is permitted.",
    "BLOCKED": "Identity, rights, integrity, verification or validity defects prohibit comparison publication."
  },
  "hardEqualityDimensions": [
    "observation kind and result-status class",
    "evaluated model/deployment identity and full system snapshot",
    "benchmark identity, revision and variant",
    "metric definition, unit, direction, aggregation and parameters",
    "evaluation harness and dataset revision",
    "prompting, decoding, scaffold, tools, budget and attempt policy",
    "sample population and denominator",
    "data cutoff",
    "integrity treatment and contamination status",
    "verification status and scope"
  ],
  "admissionBoundary": "No current score, rank or model-performance claim is admitted until an immutable observation passes schema validation, exact entity resolution, source-receipt verification, rights review and the comparison gate.",
  "normalizationBoundary": "A normalization never erases protocol differences. It must carry a source receipt, assumptions and residual limitations.",
  "rankingPolicy": "No rank is stored in the observation schema. Any future ordering must be derived from a named, reproducible view over comparison-compatible observations.",
  "apifyReceipt": "config/benchmarks/apify-research-wave84.json",
  "benchmarkProfiles": [
    {
      "id": "helm",
      "benchmarkEntityId": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "name": "HELM",
      "sourceReferences": [
        {
          "url": "https://crfm-helm.readthedocs.io/en/latest/code/",
          "locator": "RunSpec, ScenarioSpec, AdapterSpec, ExecutionSpec and MetricSpec structure",
          "evidenceRole": "identity and protocol"
        },
        {
          "url": "https://github.com/stanford-crfm/helm/blob/main/src/helm/benchmark/static/contamination.yaml",
          "locator": "contamination registry and its bounded evidence role",
          "evidenceRole": "integrity boundary"
        }
      ],
      "requiredIdentity": [
        "framework release or commit",
        "suite",
        "scenario",
        "run entry",
        "model adapter"
      ],
      "requiredProtocol": [
        "RunSpec",
        "ScenarioSpec",
        "AdapterSpec",
        "ExecutionSpec",
        "MetricSpec",
        "instance cap",
        "perturbation and seed policy"
      ],
      "comparisonBoundary": "Different suites, scenarios, adapters, prompt formats, instance caps, perturbation policies or metrics are CONTEXT_ONLY.",
      "prohibitedClaims": [
        "a bare HELM score",
        "contamination-free because the registry is silent"
      ],
      "version": "source revision 63754d05; no semantic benchmark version asserted",
      "passport": "/entities/ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589/"
    },
    {
      "id": "mlperf",
      "benchmarkEntityId": "ENT-a5bdb209-876d-458a-bd75-4b89e7dfedd6",
      "name": "MLPerf",
      "sourceReferences": [
        {
          "url": "https://github.com/mlcommons/inference_policies/blob/master/inference_rules.adoc",
          "locator": "divisions, scenarios, LoadGen, accuracy and submission rules",
          "evidenceRole": "identity and protocol"
        },
        {
          "url": "https://github.com/mlcommons/policies/blob/master/MLPerf_Compatibility_Table.adoc",
          "locator": "cross-version result compatibility rules",
          "evidenceRole": "comparison boundary"
        },
        {
          "url": "https://github.com/mlcommons/policies/blob/master/MLPerf_Results_Messaging_Guidelines.adoc",
          "locator": "public result wording and verification constraints",
          "evidenceRole": "prohibited claims"
        }
      ],
      "requiredIdentity": [
        "benchmark family and round",
        "rules commit",
        "workload",
        "division",
        "system category",
        "scenario",
        "submission ID"
      ],
      "requiredProtocol": [
        "LoadGen revision",
        "accuracy tier",
        "availability category",
        "power method",
        "system description",
        "official change state"
      ],
      "comparisonBoundary": "Round, workload, division, scenario, accuracy tier, availability and power method require an explicit compatibility statement.",
      "prohibitedClaims": [
        "unverified result described as official",
        "cross-scenario fastest-system claim"
      ],
      "version": "6.0",
      "passport": "/entities/ENT-a5bdb209-876d-458a-bd75-4b89e7dfedd6/"
    },
    {
      "id": "swe-bench",
      "benchmarkEntityId": "ENT-b2bceffe-2ce0-475c-b161-529c13553b46",
      "name": "SWE-bench",
      "sourceReferences": [
        {
          "url": "https://github.com/SWE-bench/SWE-bench/blob/main/docs/guides/evaluation.md",
          "locator": "evaluation harness, predictions, containers and grading",
          "evidenceRole": "identity and protocol"
        },
        {
          "url": "https://github.com/swe-bench/experiments/blob/main/checklist.md",
          "locator": "submission attempts, anti-solution-browsing and disclosure checklist",
          "evidenceRole": "integrity and comparison boundary"
        }
      ],
      "requiredIdentity": [
        "dataset variant and revision",
        "instance IDs",
        "repository base commits",
        "harness commit",
        "container digests"
      ],
      "requiredProtocol": [
        "scaffold commit",
        "model snapshot",
        "prompt",
        "tools and network",
        "token and time budget",
        "attempt and selection policy",
        "patch parser"
      ],
      "comparisonBoundary": "Full, Lite, Verified, Multilingual and Multimodal are distinct; best@k, pass@k and pass@1 are not interchangeable.",
      "prohibitedClaims": [
        "hidden-test access",
        "gold patch or solution browsing",
        "passed tests proves production correctness"
      ],
      "version": "Verified subset; source revision 02e7a74f",
      "passport": "/entities/ENT-b2bceffe-2ce0-475c-b161-529c13553b46/"
    },
    {
      "id": "humaneval",
      "benchmarkEntityId": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "name": "HumanEval",
      "sourceReferences": [
        {
          "url": "https://github.com/openai/human-eval/blob/master/human_eval/evaluation.py",
          "locator": "pass@k estimator and sample accounting",
          "evidenceRole": "metric contract"
        },
        {
          "url": "https://github.com/openai/human-eval/blob/master/human_eval/execution.py",
          "locator": "execution warning that reliability_guard is not a security sandbox",
          "evidenceRole": "prohibited claims"
        }
      ],
      "requiredIdentity": [
        "dataset or repository commit",
        "task bytes",
        "Python environment",
        "evaluator commit"
      ],
      "requiredProtocol": [
        "samples per task",
        "temperature and decoding",
        "k",
        "prompt format",
        "parser",
        "timeout and sandbox"
      ],
      "comparisonBoundary": "pass@k requires n, c, k and estimator details; pass@k is never silently restated as pass@1.",
      "prohibitedClaims": [
        "general software-engineering ability",
        "safe execution",
        "contamination-free"
      ],
      "version": "source revision 6d43fb98; no semantic benchmark version stated",
      "passport": "/entities/ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197/"
    },
    {
      "id": "mmlu",
      "benchmarkEntityId": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "name": "MMLU",
      "sourceReferences": [
        {
          "url": "https://github.com/hendrycks/test/blob/master/evaluate.py",
          "locator": "subject files, few-shot formatting and answer-token scoring",
          "evidenceRole": "identity and protocol"
        },
        {
          "url": "https://crfm.stanford.edu/2024/05/01/helm-mmlu.html",
          "locator": "material differences among MMLU prompting and scoring protocols",
          "evidenceRole": "comparison boundary"
        }
      ],
      "requiredIdentity": [
        "repository commit",
        "data digest",
        "exact subjects",
        "test split"
      ],
      "requiredProtocol": [
        "shots actually used",
        "demonstration IDs and order",
        "prompt and chat wrapper",
        "answer-choice order",
        "probability or generation scoring",
        "macro or micro aggregation"
      ],
      "comparisonBoundary": "Direct comparison requires identical data bytes, subjects, shots, scoring path and aggregation.",
      "prohibitedClaims": [
        "general intelligence",
        "expert-level capability without subject qualification",
        "contamination-free"
      ],
      "version": "source revision 4450500f; no semantic benchmark version stated",
      "passport": "/entities/ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70/"
    },
    {
      "id": "gpqa",
      "benchmarkEntityId": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "name": "GPQA",
      "sourceReferences": [
        {
          "url": "https://github.com/idavidrein/gpqa",
          "locator": "Main, Extended and Diamond dataset organization",
          "evidenceRole": "identity and metric"
        },
        {
          "url": "https://huggingface.co/datasets/Idavidrein/gpqa",
          "locator": "gated access, canary and non-redistribution conditions",
          "evidenceRole": "integrity and publication boundary"
        }
      ],
      "requiredIdentity": [
        "repository and data revision",
        "Main, Extended or Diamond slice",
        "item digest",
        "correction state",
        "answer-order seed"
      ],
      "requiredProtocol": [
        "prompt mode",
        "examples and chain-of-thought policy",
        "open or closed book",
        "retrieval and tools",
        "parser and abstention policy"
      ],
      "comparisonBoundary": "Main and Diamond, or retrieval and closed-book protocols, are CONTEXT_ONLY rather than silently pooled.",
      "prohibitedClaims": [
        "republishing gated question text",
        "Google-proof as a timeless guarantee"
      ],
      "version": "source revision 56686c06; no semantic benchmark version stated",
      "passport": "/entities/ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a/"
    },
    {
      "id": "arena",
      "benchmarkEntityId": "ENT-3a6be454-fedf-4237-8d71-f378e916b0a7",
      "name": "Arena",
      "sourceReferences": [
        {
          "url": "https://arena.ai/blog/policy",
          "locator": "live evaluation, vote thresholds, preliminary results and policy revisions",
          "evidenceRole": "identity and publication state"
        },
        {
          "url": "https://arxiv.org/html/2403.04132",
          "locator": "pairwise evaluation, Bradley-Terry estimation, bootstrap uncertainty and limitations",
          "evidenceRole": "metric and comparison boundary"
        }
      ],
      "requiredIdentity": [
        "product and category",
        "vote cutoff interval",
        "battle snapshot digest",
        "model aliases and endpoints",
        "pipeline and policy revisions"
      ],
      "requiredProtocol": [
        "eligibility and filtering",
        "sampling probabilities",
        "style control",
        "rating implementation",
        "bootstrap or uncertainty method",
        "battle graph"
      ],
      "comparisonBoundary": "Ratings from independently anchored snapshots or different categories and methods are not placed on one absolute scale.",
      "prohibitedClaims": [
        "rank without interval",
        "universal quality",
        "safety, factuality, latency or cost inferred from preference"
      ],
      "version": "last updated 2026-09-01",
      "passport": "/entities/ENT-3a6be454-fedf-4237-8d71-f378e916b0a7/"
    },
    {
      "id": "metr-time-horizon",
      "benchmarkEntityId": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "name": "METR Time Horizon",
      "sourceReferences": [
        {
          "url": "https://metr.org/blog/2025-03-19-measuring-ai-ability-to-complete-long-tasks/",
          "locator": "time-horizon definition, reliability target and modelling procedure",
          "evidenceRole": "metric contract"
        },
        {
          "url": "https://metr.org/blog/2024-03-15-guidelines-for-capability-elicitation/",
          "locator": "scaffold, tools, elicitation, red flags and validity reporting",
          "evidenceRole": "protocol and integrity"
        },
        {
          "url": "https://metr.org/notes/2026-01-22-time-horizon-limitations/",
          "locator": "limits on autonomy, automation and forecasting interpretations",
          "evidenceRole": "prohibited claims"
        }
      ],
      "requiredIdentity": [
        "suite and revision",
        "task IDs",
        "human-time baseline version",
        "evaluation infrastructure",
        "analysis code and model version"
      ],
      "requiredProtocol": [
        "scaffold commit",
        "tools and network",
        "elicitation procedure",
        "token and time budget",
        "run count",
        "reliability target"
      ],
      "comparisonBoundary": "Suite, task labels, infrastructure, scaffold, elicitation, budgets, reliability target and analysis model must align.",
      "prohibitedClaims": [
        "maximum autonomous duration",
        "automation percentage",
        "deployment timeline forecast"
      ],
      "version": "1.1",
      "passport": "/entities/ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e/"
    }
  ]
}
