{
  "schema": "metatron.intelligence.entity.v1",
  "id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
  "kind": "benchmark",
  "subtype": "agent_reliability_metric_suite",
  "external_identifiers": [
    {
      "scheme": "benchmark_definition_identifier",
      "value": "METR Time Horizon 1.1",
      "issuer_id": null,
      "canonical_uri": "https://metr.org/time-horizons/",
      "source": {
        "source_id": "BMRK-4",
        "url": "https://metr.org/time-horizons/"
      },
      "valid_from": null,
      "valid_to": null,
      "observed_at": "2026-09-03",
      "status": "observed"
    }
  ],
  "sources": [
    {
      "source_id": "BMRK-4",
      "url": "https://metr.org/time-horizons/",
      "title": "Task-Completion Time Horizons of Frontier AI Models",
      "publisher": "METR",
      "source_date": "2026-05-08",
      "effective_date": null,
      "retrieved_at": "2026-09-03",
      "content_sha256": "34fa9c006fae09cc0bc31d8293e61c07020a25f56c4511ffab22b5d5ae09b3ca",
      "capture_artifact": "rs8000:/opt/metatron/shared/agent-research/metatron-world-innovation-trend-atlas/raw/research/2026-09-03-wave83-benchmark-intelligence/source-captures/BMRK-4-metr-time-horizons.html",
      "capture_method": "curl --location --fail with identified browser user agent; decoded response body preserved byte-for-byte",
      "final_url": "https://metr.org/time-horizons/",
      "http_status": 200,
      "media_type": "text/html",
      "byte_count": 681146
    }
  ],
  "statements": [
    {
      "statement_id": "STM-7B5336D5FEAB0EFF6A93DF84283E3847",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "name",
      "value_or_object_id": "METR Task-Completion Time Horizon 1.1",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Version selector: `Time Horizon 1.1 (Current)`; page last updated 2026-05-08",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-2727979F9295DB8F7FBED202F6132965",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "benchmark_identifier",
      "value_or_object_id": "METR Time Horizon 1.1",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Version selector: `Time Horizon 1.1 (Current)`; page last updated 2026-05-08",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-62E1C94E892B02D55F4AFCFDE554D830",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "version",
      "value_or_object_id": "1.1",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Version selector: `Time Horizon 1.1 (Current)`; page last updated 2026-05-08",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-5503D3457D07C0670A6DA4B9389F3276",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "publisher",
      "value_or_object_id": "METR",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Version selector: `Time Horizon 1.1 (Current)`; page last updated 2026-05-08",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-B6F7AAC86BD018D070BD49CDD7D9C3D9",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "task_scope",
      "value_or_object_id": "More than one hundred self-contained software-engineering, machine-learning and cybersecurity tasks with human expert duration estimates.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Introductory definition and `Task distribution` methodological detail",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-1139542436B6F6AA7092E88034AD7FFD",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "metric_contract",
      "value_or_object_id": "Predicted human-expert task duration where an agent reaches a specified reliability, derived from a logistic success curve; the page reports 50% and 80% horizons.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Introductory definition and `Methodological Details`",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-79FCBFE4F24A7A768ECEFF708A4EB6D1",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "evaluation_conditions",
      "value_or_object_id": "Result identity includes model, scaffold, task-suite version, human-duration method, test split, reliability threshold, token/time limits, six independent task runs and any human re-scoring.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Methodological details and FAQ `What does running a time horizon evaluation involve?`",
      "valid_from": null,
      "valid_to": null,
      "confidence": "source_reported",
      "status": "observed"
    },
    {
      "statement_id": "STM-9E55C8A42578A3B0EF185DBE10AA308E",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "comparability_boundary",
      "value_or_object_id": "Compare only aligned suite version, task distribution, scaffold, duration model and reliability threshold; values above sixteen hours are marked unreliable for the current suite.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Methodological details and FAQ `What does running a time horizon evaluation involve?`",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-B60363BDC25A7666CAB5CF19400C4EE2",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "contamination_boundary",
      "value_or_object_id": "Some tasks are private, but the source does not prove that every evaluated model lacked exposure to every task or analogous task.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Methodological details and FAQ `What does running a time horizon evaluation involve?`",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-4904F9F23D6D3794A3893C99ECCACB25",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "scope",
      "value_or_object_id": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Version selector: `Time Horizon 1.1 (Current)`; page last updated 2026-05-08",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    },
    {
      "statement_id": "STM-830957FD3EB7E77C704E11E066A29FD8",
      "subject_id": "ENT-da6a449a-c99c-42d4-8e13-f5b1280fd72e",
      "predicate": "boundary",
      "value_or_object_id": "Time horizon is task difficulty measured in human time, not agent wall-clock autonomy, coverage of all intellectual work, job automation or deployment reliability.",
      "observed_at": "2026-09-03",
      "source_id": "BMRK-4",
      "source_locator": "Methodological details and FAQ `What does running a time horizon evaluation involve?`",
      "valid_from": null,
      "valid_to": null,
      "confidence": "editorial_synthesis",
      "status": "observed"
    }
  ],
  "current_projection": {
    "name": {
      "value": "METR Task-Completion Time Horizon 1.1",
      "statement_ids": [
        "STM-7B5336D5FEAB0EFF6A93DF84283E3847"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "benchmark_identifier": {
      "value": "METR Time Horizon 1.1",
      "statement_ids": [
        "STM-2727979F9295DB8F7FBED202F6132965"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "version": {
      "value": "1.1",
      "statement_ids": [
        "STM-62E1C94E892B02D55F4AFCFDE554D830"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "publisher": {
      "value": "METR",
      "statement_ids": [
        "STM-5503D3457D07C0670A6DA4B9389F3276"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "task_scope": {
      "value": "More than one hundred self-contained software-engineering, machine-learning and cybersecurity tasks with human expert duration estimates.",
      "statement_ids": [
        "STM-B6F7AAC86BD018D070BD49CDD7D9C3D9"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "metric_contract": {
      "value": "Predicted human-expert task duration where an agent reaches a specified reliability, derived from a logistic success curve; the page reports 50% and 80% horizons.",
      "statement_ids": [
        "STM-1139542436B6F6AA7092E88034AD7FFD"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "evaluation_conditions": {
      "value": "Result identity includes model, scaffold, task-suite version, human-duration method, test split, reliability threshold, token/time limits, six independent task runs and any human re-scoring.",
      "statement_ids": [
        "STM-79FCBFE4F24A7A768ECEFF708A4EB6D1"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "comparability_boundary": {
      "value": "Compare only aligned suite version, task distribution, scaffold, duration model and reliability threshold; values above sixteen hours are marked unreliable for the current suite.",
      "statement_ids": [
        "STM-9E55C8A42578A3B0EF185DBE10AA308E"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "contamination_boundary": {
      "value": "Some tasks are private, but the source does not prove that every evaluated model lacked exposure to every task or analogous task.",
      "statement_ids": [
        "STM-B60363BDC25A7666CAB5CF19400C4EE2"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "scope": {
      "value": "Reviewed benchmark or evaluation-protocol identity; model results and leaderboard ranks are separate, unrepresented observations.",
      "statement_ids": [
        "STM-4904F9F23D6D3794A3893C99ECCACB25"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "boundary": {
      "value": "Time horizon is task difficulty measured in human time, not agent wall-clock autonomy, coverage of all intellectual work, job automation or deployment reliability.",
      "statement_ids": [
        "STM-830957FD3EB7E77C704E11E066A29FD8"
      ],
      "selection_policy": "single_reviewed_statement"
    },
    "current_status": "unknown"
  },
  "claims": {
    "match_rule": "exact_source_url",
    "count": 1,
    "claim_ids": [
      "MWITA-MB-2026-007"
    ]
  },
  "relations": []
}
