{
  "schema": "metatron.intelligence.entity.v1",
  "id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
  "kind": "benchmark",
  "subtype": "graduate_science_multiple_choice_benchmark",
  "external_identifiers": [
    {
      "scheme": "benchmark_definition_identifier",
      "value": "idavidrein/gpqa@56686c06f5e19865c153de0fdb11be3890014df7",
      "issuer_id": null,
      "canonical_uri": "https://raw.githubusercontent.com/idavidrein/gpqa/56686c06f5e19865c153de0fdb11be3890014df7/README.md",
      "source": {
        "source_id": "BMRK-8",
        "url": "https://raw.githubusercontent.com/idavidrein/gpqa/56686c06f5e19865c153de0fdb11be3890014df7/README.md"
      },
      "valid_from": null,
      "valid_to": null,
      "observed_at": "2026-09-03",
      "status": "observed"
    }
  ],
  "sources": [
    {
      "source_id": "BMRK-8",
      "url": "https://raw.githubusercontent.com/idavidrein/gpqa/56686c06f5e19865c153de0fdb11be3890014df7/README.md",
      "title": "GPQA README at 56686c06",
      "publisher": "GPQA authors",
      "source_date": null,
      "effective_date": null,
      "retrieved_at": "2026-09-03",
      "content_sha256": "0be2cf410fecf82fb811f02e2314ecc5a19decc1515014498e59bae4680d4d81",
      "capture_artifact": "rs8000:/opt/metatron/shared/agent-research/metatron-world-innovation-trend-atlas/raw/research/2026-09-03-wave83-benchmark-intelligence/source-captures/BMRK-8-gpqa-readme.md",
      "capture_method": "curl --location --fail with identified browser user agent; decoded response body preserved byte-for-byte",
      "final_url": "https://raw.githubusercontent.com/idavidrein/gpqa/56686c06f5e19865c153de0fdb11be3890014df7/README.md",
      "http_status": 200,
      "media_type": "text/plain",
      "byte_count": 3557
    }
  ],
  "statements": [
    {
      "statement_id": "STM-F8F7511C4887A7F5770F4930E8FF0FC2",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "name",
      "value_or_object_id": "GPQA",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README heading, paper citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-D7938E87D03CE93AE550F9C3F6420755",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "benchmark_identifier",
      "value_or_object_id": "idavidrein/gpqa@56686c06f5e19865c153de0fdb11be3890014df7",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README heading, paper citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-3D118B062230149A5FE70BB2C8AD8ABC",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "version",
      "value_or_object_id": "source revision 56686c06; no semantic benchmark version stated",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README heading, paper citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-C05BEC6D673C202E03FBC8079A1F1D68",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "publisher",
      "value_or_object_id": "GPQA authors",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README heading, paper citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-349D2381D104EDF2254D5F6DD08C3AD1",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "task_scope",
      "value_or_object_id": "Graduate-level, Google-proof question-answering dataset with separate data files and baseline implementations.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README heading, dataset download and baseline usage",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-E299363DB1D9B97B5AEBE4C43032A4A2",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "metric_contract",
      "value_or_object_id": "A result is meaningful only with the exact data file/subset, shuffled answer-choice seed and prompt/retrieval mode; no single aggregate is admitted here.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README CLI arguments for data filename, prompt type and seed",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-5A03BC625D23E7C900CD218390ED3652",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "evaluation_conditions",
      "value_or_object_id": "Result identity includes repository/data revision, subset, model, zero/few-shot or chain-of-thought mode, closed/open-book retrieval setting, answer-choice shuffle seed and cache behavior.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README baseline arguments and closed/open-book commands",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-963A16D9BE8FCA3CF04A849FDD3CBB08",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "comparability_boundary",
      "value_or_object_id": "Do not combine GPQA subsets or compare retrieval and closed-book runs, different prompts, seeds or code revisions as one measurement.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README baseline arguments and closed/open-book commands",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-50D5F50D1EDDC23F522A32855BFF27D2",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "contamination_boundary",
      "value_or_object_id": "The dataset includes a canary string; this is a detection/deterrence aid, not proof that training exposure did not occur.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README baseline arguments and closed/open-book commands",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-59F6D590D7EFF76C80661ED51F1C5F51",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "scope",
      "value_or_object_id": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README heading, paper citation and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-C09AD5EBC096AEB96DB01B236BB3CA4D",
      "subject_id": "ENT-9a3c06c8-568b-4f39-9f62-e1edb55e8a6a",
      "predicate": "boundary",
      "value_or_object_id": "The `Google-Proof` title is the benchmark name, not a guarantee that every item is unsearchable, uncontaminated or representative of all expert reasoning.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-8",
      "source_locator": "README baseline arguments and closed/open-book commands",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    }
  ],
  "current_projection": {
    "name": {
      "value": "GPQA",
      "statement_ids": [
        "STM-F8F7511C4887A7F5770F4930E8FF0FC2"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "benchmark_identifier": {
      "value": "idavidrein/gpqa@56686c06f5e19865c153de0fdb11be3890014df7",
      "statement_ids": [
        "STM-D7938E87D03CE93AE550F9C3F6420755"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "version": {
      "value": "source revision 56686c06; no semantic benchmark version stated",
      "statement_ids": [
        "STM-3D118B062230149A5FE70BB2C8AD8ABC"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "publisher": {
      "value": "GPQA authors",
      "statement_ids": [
        "STM-C05BEC6D673C202E03FBC8079A1F1D68"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "task_scope": {
      "value": "Graduate-level, Google-proof question-answering dataset with separate data files and baseline implementations.",
      "statement_ids": [
        "STM-349D2381D104EDF2254D5F6DD08C3AD1"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "metric_contract": {
      "value": "A result is meaningful only with the exact data file/subset, shuffled answer-choice seed and prompt/retrieval mode; no single aggregate is admitted here.",
      "statement_ids": [
        "STM-E299363DB1D9B97B5AEBE4C43032A4A2"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "evaluation_conditions": {
      "value": "Result identity includes repository/data revision, subset, model, zero/few-shot or chain-of-thought mode, closed/open-book retrieval setting, answer-choice shuffle seed and cache behavior.",
      "statement_ids": [
        "STM-5A03BC625D23E7C900CD218390ED3652"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "comparability_boundary": {
      "value": "Do not combine GPQA subsets or compare retrieval and closed-book runs, different prompts, seeds or code revisions as one measurement.",
      "statement_ids": [
        "STM-963A16D9BE8FCA3CF04A849FDD3CBB08"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "contamination_boundary": {
      "value": "The dataset includes a canary string; this is a detection/deterrence aid, not proof that training exposure did not occur.",
      "statement_ids": [
        "STM-50D5F50D1EDDC23F522A32855BFF27D2"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "scope": {
      "value": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "statement_ids": [
        "STM-59F6D590D7EFF76C80661ED51F1C5F51"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "boundary": {
      "value": "The `Google-Proof` title is the benchmark name, not a guarantee that every item is unsearchable, uncontaminated or representative of all expert reasoning.",
      "statement_ids": [
        "STM-C09AD5EBC096AEB96DB01B236BB3CA4D"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "current_status": "unknown"
  },
  "claims": {
    "match_rule": "exact_source_url",
    "count": 0,
    "claim_ids": []
  },
  "relations": []
}
