{
  "schema": "metatron.intelligence.entity.v1",
  "id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
  "kind": "benchmark",
  "subtype": "multi_task_language_model_benchmark",
  "external_identifiers": [
    {
      "scheme": "benchmark_definition_identifier",
      "value": "google/BIG-bench@092b196c1f8f14a54bbc62f24759d43bde46dd3b",
      "issuer_id": null,
      "canonical_uri": "https://raw.githubusercontent.com/google/BIG-bench/092b196c1f8f14a54bbc62f24759d43bde46dd3b/README.md",
      "source": {
        "source_id": "BMRK-7",
        "url": "https://raw.githubusercontent.com/google/BIG-bench/092b196c1f8f14a54bbc62f24759d43bde46dd3b/README.md"
      },
      "valid_from": null,
      "valid_to": null,
      "observed_at": "2026-09-03",
      "status": "observed"
    }
  ],
  "sources": [
    {
      "source_id": "BMRK-7",
      "url": "https://raw.githubusercontent.com/google/BIG-bench/092b196c1f8f14a54bbc62f24759d43bde46dd3b/README.md",
      "title": "BIG-bench README at 092b196c",
      "publisher": "BIG-bench authors / Google repository",
      "source_date": null,
      "effective_date": null,
      "retrieved_at": "2026-09-03",
      "content_sha256": "be15e44863b6070a1dda2000d4ba16b89e8d82208cae92b07a3ab6531e885a12",
      "capture_artifact": "rs8000:/opt/metatron/shared/agent-research/metatron-world-innovation-trend-atlas/raw/research/2026-09-03-wave83-benchmark-intelligence/source-captures/BMRK-7-big-bench-readme.md",
      "capture_method": "curl --location --fail with identified browser user agent; decoded response body preserved byte-for-byte",
      "final_url": "https://raw.githubusercontent.com/google/BIG-bench/092b196c1f8f14a54bbc62f24759d43bde46dd3b/README.md",
      "http_status": 200,
      "media_type": "text/plain",
      "byte_count": 18767
    }
  ],
  "statements": [
    {
      "statement_id": "STM-8B431ED7CC9B45EE8F701FDB4EFB861F",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "name",
      "value_or_object_id": "BIG-bench",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README heading, benchmark description and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-1375BEB1A2E8DF0A8D92555221313579",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "benchmark_identifier",
      "value_or_object_id": "google/BIG-bench@092b196c1f8f14a54bbc62f24759d43bde46dd3b",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README heading, benchmark description and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-1E7F321A3DD469732CBF75C72E507625",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "version",
      "value_or_object_id": "source revision 092b196c; no semantic benchmark version stated",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README heading, benchmark description and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-9EAB8915C865F1990384E772CB94EBF3",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "publisher",
      "value_or_object_id": "BIG-bench authors",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README heading, benchmark description and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-050EDD835250DE41DE54A4B9E0A3B525",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "task_scope",
      "value_or_object_id": "Collaborative suite of more than 200 language-model tasks; BIG-bench Lite is a distinct 24-task subset.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README introduction and `BIG-bench Lite leaderboard`",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-A169695C9260F14A6A4E0517C3711CD5",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "metric_contract",
      "value_or_object_id": "Each task declares its metrics and preferred score; programmatic and JSON tasks can use different evaluation functions, so no universal raw score is represented.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README task metadata and `preferred_score` documentation",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-8068736D536A934B0BFC50E09678F5EC",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "evaluation_conditions",
      "value_or_object_id": "Result identity requires exact repository/task set, full versus Lite selection, task definitions, metric, model interface, shot count and evaluation code.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README task creation, evaluation and result-submission sections",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-EEA17B6963A9267A47B4A9D8C02369FA",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "comparability_boundary",
      "value_or_object_id": "BIG-bench full, BIG-bench Lite and individual tasks are not interchangeable; comparisons require the same task set, revision, prompting and preferred metrics.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README task creation, evaluation and result-submission sections",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-1D53E02E4F95E4275D142A2C595B763B",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "contamination_boundary",
      "value_or_object_id": "Task files carry an explicit canary intended to deter inclusion in web-scraped training corpora; a canary does not prove absence of exposure.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README task creation, evaluation and result-submission sections",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-F527E7BE2953CB365245567C1EDA73B0",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "scope",
      "value_or_object_id": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README heading, benchmark description and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-8BF75AAEAF2D350F98E29B22FBE1AE3B",
      "subject_id": "ENT-232ec547-65a7-4358-b573-0445af54153a",
      "predicate": "boundary",
      "value_or_object_id": "Community task breadth does not imply complete capability coverage, and leaderboard results are not represented by this identity record.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-7",
      "source_locator": "README task creation, evaluation and result-submission sections",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    }
  ],
  "current_projection": {
    "name": {
      "value": "BIG-bench",
      "statement_ids": [
        "STM-8B431ED7CC9B45EE8F701FDB4EFB861F"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "benchmark_identifier": {
      "value": "google/BIG-bench@092b196c1f8f14a54bbc62f24759d43bde46dd3b",
      "statement_ids": [
        "STM-1375BEB1A2E8DF0A8D92555221313579"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "version": {
      "value": "source revision 092b196c; no semantic benchmark version stated",
      "statement_ids": [
        "STM-1E7F321A3DD469732CBF75C72E507625"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "publisher": {
      "value": "BIG-bench authors",
      "statement_ids": [
        "STM-9EAB8915C865F1990384E772CB94EBF3"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "task_scope": {
      "value": "Collaborative suite of more than 200 language-model tasks; BIG-bench Lite is a distinct 24-task subset.",
      "statement_ids": [
        "STM-050EDD835250DE41DE54A4B9E0A3B525"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "metric_contract": {
      "value": "Each task declares its metrics and preferred score; programmatic and JSON tasks can use different evaluation functions, so no universal raw score is represented.",
      "statement_ids": [
        "STM-A169695C9260F14A6A4E0517C3711CD5"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "evaluation_conditions": {
      "value": "Result identity requires exact repository/task set, full versus Lite selection, task definitions, metric, model interface, shot count and evaluation code.",
      "statement_ids": [
        "STM-8068736D536A934B0BFC50E09678F5EC"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "comparability_boundary": {
      "value": "BIG-bench full, BIG-bench Lite and individual tasks are not interchangeable; comparisons require the same task set, revision, prompting and preferred metrics.",
      "statement_ids": [
        "STM-EEA17B6963A9267A47B4A9D8C02369FA"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "contamination_boundary": {
      "value": "Task files carry an explicit canary intended to deter inclusion in web-scraped training corpora; a canary does not prove absence of exposure.",
      "statement_ids": [
        "STM-1D53E02E4F95E4275D142A2C595B763B"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "scope": {
      "value": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "statement_ids": [
        "STM-F527E7BE2953CB365245567C1EDA73B0"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "boundary": {
      "value": "Community task breadth does not imply complete capability coverage, and leaderboard results are not represented by this identity record.",
      "statement_ids": [
        "STM-8BF75AAEAF2D350F98E29B22FBE1AE3B"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "current_status": "unknown"
  },
  "claims": {
    "match_rule": "exact_source_url",
    "count": 0,
    "claim_ids": []
  },
  "relations": []
}
