{
  "schema": "metatron.intelligence.entity.v1",
  "id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
  "kind": "benchmark",
  "subtype": "code_generation_functional_correctness_benchmark",
  "external_identifiers": [
    {
      "scheme": "benchmark_definition_identifier",
      "value": "openai/human-eval@6d43fb980f9fee3c892a914eda09951f772ad10d",
      "issuer_id": null,
      "canonical_uri": "https://raw.githubusercontent.com/openai/human-eval/6d43fb980f9fee3c892a914eda09951f772ad10d/README.md",
      "source": {
        "source_id": "BMRK-6",
        "url": "https://raw.githubusercontent.com/openai/human-eval/6d43fb980f9fee3c892a914eda09951f772ad10d/README.md"
      },
      "valid_from": null,
      "valid_to": null,
      "observed_at": "2026-09-03",
      "status": "observed"
    }
  ],
  "sources": [
    {
      "source_id": "BMRK-6",
      "url": "https://raw.githubusercontent.com/openai/human-eval/6d43fb980f9fee3c892a914eda09951f772ad10d/README.md",
      "title": "HumanEval README at 6d43fb98",
      "publisher": "OpenAI",
      "source_date": null,
      "effective_date": null,
      "retrieved_at": "2026-09-03",
      "content_sha256": "2649c7f6b03c8aecdc480e4f25333b113e4d89bf819ba911e4a7983adabd12ce",
      "capture_artifact": "rs8000:/opt/metatron/shared/agent-research/metatron-world-innovation-trend-atlas/raw/research/2026-09-03-wave83-benchmark-intelligence/source-captures/BMRK-6-human-eval-readme.md",
      "capture_method": "curl --location --fail with identified browser user agent; decoded response body preserved byte-for-byte",
      "final_url": "https://raw.githubusercontent.com/openai/human-eval/6d43fb980f9fee3c892a914eda09951f772ad10d/README.md",
      "http_status": 200,
      "media_type": "text/plain",
      "byte_count": 4848
    }
  ],
  "statements": [
    {
      "statement_id": "STM-B7E30B05650C36DC6DE46A1D37DD9103",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "name",
      "value_or_object_id": "HumanEval",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README heading and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-A1BF0D74D592F78A807AE0375D1DDC7F",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "benchmark_identifier",
      "value_or_object_id": "openai/human-eval@6d43fb980f9fee3c892a914eda09951f772ad10d",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README heading and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-CBB55EC5B0FB0664EC54E0CEB7047F08",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "version",
      "value_or_object_id": "source revision 6d43fb98; no semantic benchmark version stated",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README heading and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-DE2666F65F9BAAA255CD014533202D47",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "publisher",
      "value_or_object_id": "OpenAI",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README heading and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-D0AA862DC58A8429137C0435F1CF88BC",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "task_scope",
      "value_or_object_id": "Hand-written problem-solving dataset for generated Python function completions evaluated by executable tests.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README introduction, sample format and functional-correctness evaluator",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-5EE6112402B7B9390CE2E1B9C28DD148",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "metric_contract",
      "value_or_object_id": "pass@k estimated from multiple generated samples; the official evaluator refuses cases with fewer samples than k because no unbiased estimator is available.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README evaluator output and pass@k limitation",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-491C2B919F6A89D125BC14C07F39F64D",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "evaluation_conditions",
      "value_or_object_id": "Result identity includes problem file, sample count per task, k values, generation settings, completion format, evaluator revision and sandbox/runtime resources.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README usage, JSONL format, evaluator options, safety warning and known issues",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-523CFD44015B7F7015E98435E454E710",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "comparability_boundary",
      "value_or_object_id": "Compare only results with aligned task data, sampling count and generation setup, k, evaluator revision and runtime; pass@1 and pass@100 answer different questions.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README usage, JSONL format, evaluator options, safety warning and known issues",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-2A43108080BC1AC98D2DE1E1341AC699",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "contamination_boundary",
      "value_or_object_id": "The selected source does not establish absence of HumanEval problem or solution exposure in training data.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README usage, JSONL format, evaluator options, safety warning and known issues",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-B1DB8AE5E665CEB002F70674BFC63787",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "scope",
      "value_or_object_id": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README heading and pinned repository revision",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-8E641B146AE449FE5E7928D4042738BD",
      "subject_id": "ENT-3e5d2087-d0b9-4f35-b4a7-dcd41c022197",
      "predicate": "boundary",
      "value_or_object_id": "Executing generated code is explicitly unsafe without a robust sandbox. Functional-test passage is not proof of secure, maintainable or production-ready code.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-6",
      "source_locator": "README usage, JSONL format, evaluator options, safety warning and known issues",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    }
  ],
  "current_projection": {
    "name": {
      "value": "HumanEval",
      "statement_ids": [
        "STM-B7E30B05650C36DC6DE46A1D37DD9103"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "benchmark_identifier": {
      "value": "openai/human-eval@6d43fb980f9fee3c892a914eda09951f772ad10d",
      "statement_ids": [
        "STM-A1BF0D74D592F78A807AE0375D1DDC7F"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "version": {
      "value": "source revision 6d43fb98; no semantic benchmark version stated",
      "statement_ids": [
        "STM-CBB55EC5B0FB0664EC54E0CEB7047F08"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "publisher": {
      "value": "OpenAI",
      "statement_ids": [
        "STM-DE2666F65F9BAAA255CD014533202D47"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "task_scope": {
      "value": "Hand-written problem-solving dataset for generated Python function completions evaluated by executable tests.",
      "statement_ids": [
        "STM-D0AA862DC58A8429137C0435F1CF88BC"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "metric_contract": {
      "value": "pass@k estimated from multiple generated samples; the official evaluator refuses cases with fewer samples than k because no unbiased estimator is available.",
      "statement_ids": [
        "STM-5EE6112402B7B9390CE2E1B9C28DD148"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "evaluation_conditions": {
      "value": "Result identity includes problem file, sample count per task, k values, generation settings, completion format, evaluator revision and sandbox/runtime resources.",
      "statement_ids": [
        "STM-491C2B919F6A89D125BC14C07F39F64D"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "comparability_boundary": {
      "value": "Compare only results with aligned task data, sampling count and generation setup, k, evaluator revision and runtime; pass@1 and pass@100 answer different questions.",
      "statement_ids": [
        "STM-523CFD44015B7F7015E98435E454E710"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "contamination_boundary": {
      "value": "The selected source does not establish absence of HumanEval problem or solution exposure in training data.",
      "statement_ids": [
        "STM-2A43108080BC1AC98D2DE1E1341AC699"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "scope": {
      "value": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "statement_ids": [
        "STM-B1DB8AE5E665CEB002F70674BFC63787"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "boundary": {
      "value": "Executing generated code is explicitly unsafe without a robust sandbox. Functional-test passage is not proof of secure, maintainable or production-ready code.",
      "statement_ids": [
        "STM-8E641B146AE449FE5E7928D4042738BD"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "current_status": "unknown"
  },
  "claims": {
    "match_rule": "exact_source_url",
    "count": 0,
    "claim_ids": []
  },
  "relations": []
}
