{
  "schema": "metatron.intelligence.entity.v1",
  "id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
  "kind": "benchmark",
  "subtype": "multidomain_multiple_choice_benchmark",
  "external_identifiers": [
    {
      "scheme": "benchmark_definition_identifier",
      "value": "hendrycks/test@4450500f923c49f1fb1dd3d99108a0bd9717b660",
      "issuer_id": null,
      "canonical_uri": "https://raw.githubusercontent.com/hendrycks/test/4450500f923c49f1fb1dd3d99108a0bd9717b660/README.md",
      "source": {
        "source_id": "BMRK-9",
        "url": "https://raw.githubusercontent.com/hendrycks/test/4450500f923c49f1fb1dd3d99108a0bd9717b660/README.md"
      },
      "valid_from": null,
      "valid_to": null,
      "observed_at": "2026-09-03",
      "status": "observed"
    }
  ],
  "sources": [
    {
      "source_id": "BMRK-9",
      "url": "https://raw.githubusercontent.com/hendrycks/test/4450500f923c49f1fb1dd3d99108a0bd9717b660/README.md",
      "title": "MMLU README at 4450500f",
      "publisher": "MMLU authors",
      "source_date": null,
      "effective_date": null,
      "retrieved_at": "2026-09-03",
      "content_sha256": "59e7d8e808141d3eb8ba02bf38a53db6237d73cbe91d4d823693ead1f8f75182",
      "capture_artifact": "rs8000:/opt/metatron/shared/agent-research/metatron-world-innovation-trend-atlas/raw/research/2026-09-03-wave83-benchmark-intelligence/source-captures/BMRK-9-mmlu-readme.md",
      "capture_method": "curl --location --fail with identified browser user agent; decoded response body preserved byte-for-byte",
      "final_url": "https://raw.githubusercontent.com/hendrycks/test/4450500f923c49f1fb1dd3d99108a0bd9717b660/README.md",
      "http_status": 200,
      "media_type": "text/plain",
      "byte_count": 3256
    }
  ],
  "statements": [
    {
      "statement_id": "STM-9A08F6827C66620675E795F9C6EE5DFE",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "name",
      "value_or_object_id": "Massive Multitask Language Understanding (MMLU)",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README heading, ICLR 2021 citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-F2AC00F7067279598E52D970C4180575",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "benchmark_identifier",
      "value_or_object_id": "hendrycks/test@4450500f923c49f1fb1dd3d99108a0bd9717b660",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README heading, ICLR 2021 citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-CF5C796C70E5B2BEA8340C063B3A5FCE",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "version",
      "value_or_object_id": "source revision 4450500f; no semantic benchmark version stated",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README heading, ICLR 2021 citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-C216D2E086B67B64504D3F7B1F36E31B",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "publisher",
      "value_or_object_id": "MMLU authors",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README heading, ICLR 2021 citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-FE2D01CDBC2EA67702A042238FE4BBB7",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "task_scope",
      "value_or_object_id": "Multitask language-understanding test whose published category grouping includes humanities, social sciences, STEM and other subjects.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README title, test download and leaderboard category headings",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-32EDD6D2D9615EA20109AE2DBF728406",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "metric_contract",
      "value_or_object_id": "Category results and an average are reported for the cited test, but prompt and shot conditions remain part of each model result rather than the benchmark identity.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README test leaderboard columns and model-condition labels",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-0ABCC20925BA405C957BCB7181DBCAFD",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "evaluation_conditions",
      "value_or_object_id": "Result identity requires exact test data, repository/evaluation-code revision, model version, prompting and shot/fine-tuning condition, category aggregation and scoring implementation.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README repository/test description and leaderboard condition labels",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-D9B1AE94D5BD0E47DF5AEEA37FA8F2BF",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "comparability_boundary",
      "value_or_object_id": "Do not treat results with different prompts, shot counts, fine-tuning, model snapshots, data revisions or category aggregation as directly comparable.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README repository/test description and leaderboard condition labels",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-6076239926FCEC468695B6BDE028D428",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "contamination_boundary",
      "value_or_object_id": "The selected source does not establish absence of test-item exposure in model training or tuning data.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README repository/test description and leaderboard condition labels",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-514AEEF6C207392406EE9CBB8E54C72D",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "scope",
      "value_or_object_id": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README heading, ICLR 2021 citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-A6DFD372C0AE010C0B2DF5C9A0C06C23",
      "subject_id": "ENT-c2a9f450-18f3-47d2-ac6f-8e709b396a70",
      "predicate": "boundary",
      "value_or_object_id": "MMLU measures performance on its selected test and categories; it is not a universal measure of intelligence, reasoning or deployment fitness.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-9",
      "source_locator": "README repository/test description and leaderboard condition labels",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    }
  ],
  "current_projection": {
    "name": {
      "value": "Massive Multitask Language Understanding (MMLU)",
      "statement_ids": [
        "STM-9A08F6827C66620675E795F9C6EE5DFE"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "benchmark_identifier": {
      "value": "hendrycks/test@4450500f923c49f1fb1dd3d99108a0bd9717b660",
      "statement_ids": [
        "STM-F2AC00F7067279598E52D970C4180575"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "version": {
      "value": "source revision 4450500f; no semantic benchmark version stated",
      "statement_ids": [
        "STM-CF5C796C70E5B2BEA8340C063B3A5FCE"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "publisher": {
      "value": "MMLU authors",
      "statement_ids": [
        "STM-C216D2E086B67B64504D3F7B1F36E31B"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "task_scope": {
      "value": "Multitask language-understanding test whose published category grouping includes humanities, social sciences, STEM and other subjects.",
      "statement_ids": [
        "STM-FE2D01CDBC2EA67702A042238FE4BBB7"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "metric_contract": {
      "value": "Category results and an average are reported for the cited test, but prompt and shot conditions remain part of each model result rather than the benchmark identity.",
      "statement_ids": [
        "STM-32EDD6D2D9615EA20109AE2DBF728406"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "evaluation_conditions": {
      "value": "Result identity requires exact test data, repository/evaluation-code revision, model version, prompting and shot/fine-tuning condition, category aggregation and scoring implementation.",
      "statement_ids": [
        "STM-0ABCC20925BA405C957BCB7181DBCAFD"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "comparability_boundary": {
      "value": "Do not treat results with different prompts, shot counts, fine-tuning, model snapshots, data revisions or category aggregation as directly comparable.",
      "statement_ids": [
        "STM-D9B1AE94D5BD0E47DF5AEEA37FA8F2BF"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "contamination_boundary": {
      "value": "The selected source does not establish absence of test-item exposure in model training or tuning data.",
      "statement_ids": [
        "STM-6076239926FCEC468695B6BDE028D428"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "scope": {
      "value": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "statement_ids": [
        "STM-514AEEF6C207392406EE9CBB8E54C72D"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "boundary": {
      "value": "MMLU measures performance on its selected test and categories; it is not a universal measure of intelligence, reasoning or deployment fitness.",
      "statement_ids": [
        "STM-A6DFD372C0AE010C0B2DF5C9A0C06C23"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "current_status": "unknown"
  },
  "claims": {
    "match_rule": "exact_source_url",
    "count": 0,
    "claim_ids": []
  },
  "relations": []
}
