{
  "schema": "metatron.intelligence.entity.v1",
  "id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
  "kind": "benchmark",
  "subtype": "evaluation_framework_revision",
  "external_identifiers": [
    {
      "scheme": "benchmark_definition_identifier",
      "value": "stanford-crfm/helm@63754d05db6f874e41a395880fb573890a13e791",
      "issuer_id": null,
      "canonical_uri": "https://raw.githubusercontent.com/stanford-crfm/helm/63754d05db6f874e41a395880fb573890a13e791/README.md",
      "source": {
        "source_id": "BMRK-1",
        "url": "https://raw.githubusercontent.com/stanford-crfm/helm/63754d05db6f874e41a395880fb573890a13e791/README.md"
      },
      "valid_from": null,
      "valid_to": null,
      "observed_at": "2026-09-03",
      "status": "observed"
    }
  ],
  "sources": [
    {
      "source_id": "BMRK-1",
      "url": "https://raw.githubusercontent.com/stanford-crfm/helm/63754d05db6f874e41a395880fb573890a13e791/README.md",
      "title": "HELM README at 63754d05",
      "publisher": "Stanford CRFM",
      "source_date": null,
      "effective_date": null,
      "retrieved_at": "2026-09-03",
      "content_sha256": "465390a907127138096869b49fd79c817392f5404d23d98161d881f2e408a303",
      "capture_artifact": "rs8000:/opt/metatron/shared/agent-research/metatron-world-innovation-trend-atlas/raw/research/2026-09-03-wave83-benchmark-intelligence/source-captures/BMRK-1-helm-readme.md",
      "capture_method": "curl --location --fail with identified browser user agent; decoded response body preserved byte-for-byte",
      "final_url": "https://raw.githubusercontent.com/stanford-crfm/helm/63754d05db6f874e41a395880fb573890a13e791/README.md",
      "http_status": 200,
      "media_type": "text/plain",
      "byte_count": 7194
    }
  ],
  "statements": [
    {
      "statement_id": "STM-8F038EC18C867E67D693E1B558EBA633",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "name",
      "value_or_object_id": "Holistic Evaluation of Language Models (HELM)",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README heading, creation statement and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-6681D296DF1ED6A1DE500F79FAB39B22",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "benchmark_identifier",
      "value_or_object_id": "stanford-crfm/helm@63754d05db6f874e41a395880fb573890a13e791",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README heading, creation statement and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-173C37511B89DD20D1D19B395DB99B76",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "version",
      "value_or_object_id": "source revision 63754d05; no semantic benchmark version asserted",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README heading, creation statement and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-7D835A2DAF7F5B16FDEC67EFE24F1CB5",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "publisher",
      "value_or_object_id": "Stanford CRFM",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README heading, creation statement and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-6C93E60D2501DDC0C15B1372E6340E81",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "task_scope",
      "value_or_object_id": "Framework for standardized, reproducible evaluation of language and multimodal foundation models across datasets and benchmarks.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README feature list: standardized datasets/benchmarks and unified model interface",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-ED2235DEBA1B041DFE2094589BF0DA91",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "metric_contract",
      "value_or_object_id": "Scenario-specific metrics include dimensions beyond accuracy, such as efficiency, bias and toxicity; no single HELM score is represented.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README feature list: metrics beyond accuracy",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-0D184EF2537905737B098E6E7A5403E7",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "evaluation_conditions",
      "value_or_object_id": "A result is conditioned on the pinned framework revision, run entry, suite, model adapter, scenario and metric configuration.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README Quick Start and configuration links",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-E12CA160071807C9621B5A442FDB79D6",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "comparability_boundary",
      "value_or_object_id": "Compare only runs with aligned HELM revision, scenario, adapter, prompting/run specification and metric; the `latest` route is not a fixed version.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README Quick Start and configuration links",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-E2EF1619EA57493F4DE430FE652E038B",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "contamination_boundary",
      "value_or_object_id": "The selected source does not establish absence of benchmark exposure in model training data.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README Quick Start and configuration links",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-DB0E5854303186E131F766C58A9EA007",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "scope",
      "value_or_object_id": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README heading, creation statement and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-21402004F1D6DFC43D6F9BAF95CCA4F7",
      "subject_id": "ENT-ff3889bc-c073-43cd-b5a3-2b5aa262e589",
      "predicate": "boundary",
      "value_or_object_id": "HELM entered maintenance mode on 2026-06-01. Framework availability is not evidence that every hosted leaderboard result is current or comparable.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-1",
      "source_locator": "README Quick Start and configuration links",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    }
  ],
  "current_projection": {
    "name": {
      "value": "Holistic Evaluation of Language Models (HELM)",
      "statement_ids": [
        "STM-8F038EC18C867E67D693E1B558EBA633"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "benchmark_identifier": {
      "value": "stanford-crfm/helm@63754d05db6f874e41a395880fb573890a13e791",
      "statement_ids": [
        "STM-6681D296DF1ED6A1DE500F79FAB39B22"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "version": {
      "value": "source revision 63754d05; no semantic benchmark version asserted",
      "statement_ids": [
        "STM-173C37511B89DD20D1D19B395DB99B76"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "publisher": {
      "value": "Stanford CRFM",
      "statement_ids": [
        "STM-7D835A2DAF7F5B16FDEC67EFE24F1CB5"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "task_scope": {
      "value": "Framework for standardized, reproducible evaluation of language and multimodal foundation models across datasets and benchmarks.",
      "statement_ids": [
        "STM-6C93E60D2501DDC0C15B1372E6340E81"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "metric_contract": {
      "value": "Scenario-specific metrics include dimensions beyond accuracy, such as efficiency, bias and toxicity; no single HELM score is represented.",
      "statement_ids": [
        "STM-ED2235DEBA1B041DFE2094589BF0DA91"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "evaluation_conditions": {
      "value": "A result is conditioned on the pinned framework revision, run entry, suite, model adapter, scenario and metric configuration.",
      "statement_ids": [
        "STM-0D184EF2537905737B098E6E7A5403E7"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "comparability_boundary": {
      "value": "Compare only runs with aligned HELM revision, scenario, adapter, prompting/run specification and metric; the `latest` route is not a fixed version.",
      "statement_ids": [
        "STM-E12CA160071807C9621B5A442FDB79D6"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "contamination_boundary": {
      "value": "The selected source does not establish absence of benchmark exposure in model training data.",
      "statement_ids": [
        "STM-E2EF1619EA57493F4DE430FE652E038B"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "scope": {
      "value": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "statement_ids": [
        "STM-DB0E5854303186E131F766C58A9EA007"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "boundary": {
      "value": "HELM entered maintenance mode on 2026-06-01. Framework availability is not evidence that every hosted leaderboard result is current or comparable.",
      "statement_ids": [
        "STM-21402004F1D6DFC43D6F9BAF95CCA4F7"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "current_status": "unknown"
  },
  "claims": {
    "match_rule": "exact_source_url",
    "count": 0,
    "claim_ids": []
  },
  "relations": []
}
