{
  "schemaVersion": 1,
  "moduleId": "agent-horizon",
  "generatedAt": "2026-08-23T00:52:22.230Z",
  "evidenceState": {
    "claimId": "agent-horizon/claim/evidence-state",
    "lastSourceCheckAt": "2026-08-23",
    "maturity": {
      "level": "developing",
      "label": "UTVIKLENDE",
      "rationale": "4 kildebundne P50/P80-estimater med intervaller i samme TH1.1-kontrakt; suitedekning og 16-timersgrensen begrenser fortsatt overførbarheten."
    },
    "cadence": "monthly",
    "compatibleObservationCount": 4,
    "compatibleObservationLabel": "ikke-kontekstuelle P50/P80-estimater i samme TH1.1-kontrakt",
    "trend": {
      "status": "not-established",
      "label": "IKKE ETABLERT",
      "rationale": "4 estimater kan sammenlignes på tvers av modell og reliabilitetsnivå, men er ikke en tidsserie; Opus 4.5 vises bare som kontekst."
    },
    "measurementGaps": [
      {
        "label": "oppgaver over 16 timer",
        "claimId": "agent-horizon/claim/long-task-limit"
      },
      {
        "label": "økonomisk overførbarhet",
        "claimId": "agent-horizon/claim/suite-coverage"
      }
    ]
  },
  "snapshot": {
    "schemaVersion": 1,
    "moduleId": "agent-horizon",
    "title": "Agent Task Horizon",
    "eyebrow": "KAPABILITET / PÅLITELIGHET",
    "asOf": "2026-05-08",
    "retrievedAt": "2026-08-23",
    "generatedAt": "2026-08-23T00:52:22.230Z",
    "sourceIds": [
      "metr-time-horizon-1-1-2026",
      "metr-time-horizons-2026",
      "metr-frontier-risk-report-2026",
      "agent-horizon-metr-th11-results-aae31902b051"
    ],
    "status": "sparse",
    "evidenceState": {
      "claimId": "agent-horizon/claim/evidence-state",
      "lastSourceCheckAt": "2026-08-23",
      "maturity": {
        "level": "developing",
        "label": "UTVIKLENDE",
        "rationale": "4 kildebundne P50/P80-estimater med intervaller i samme TH1.1-kontrakt; suitedekning og 16-timersgrensen begrenser fortsatt overførbarheten."
      },
      "cadence": "monthly",
      "compatibleObservationCount": 4,
      "compatibleObservationLabel": "ikke-kontekstuelle P50/P80-estimater i samme TH1.1-kontrakt",
      "trend": {
        "status": "not-established",
        "label": "IKKE ETABLERT",
        "rationale": "4 estimater kan sammenlignes på tvers av modell og reliabilitetsnivå, men er ikke en tidsserie; Opus 4.5 vises bare som kontekst."
      },
      "measurementGaps": [
        {
          "label": "oppgaver over 16 timer",
          "claimId": "agent-horizon/claim/long-task-limit"
        },
        {
          "label": "økonomisk overførbarhet",
          "claimId": "agent-horizon/claim/suite-coverage"
        }
      ]
    },
    "decisionQuestion": "Hvor lange programvareoppgaver klarer frontier-agenter når vi eksplisitt velger suksesskrav og respekterer suitedekningen?",
    "summary": "Offentlig frontier: ca. 12 t ved 50 % suksess, men ca. 1,5 t ved 80 %",
    "valueStatement": "Se hvor mye autonomihorisonten krymper når pålitelighetskravet øker—og når benchmarken går tom for bevis.",
    "valueClaimId": "agent-horizon/claim/reliability-gap",
    "analysisRoute": "/tracker/agent-horizon/reliability",
    "measurements": [
      {
        "id": "agent-horizon/frontier-2026/public-p50",
        "label": "Offentlig frontier, 50 % horisont",
        "value": "ca. 12 t [5–61]",
        "detail": "METRs Feb–Mar 2026-vurdering; bredt intervall og suite nær metning.",
        "classification": "observation",
        "claimId": "agent-horizon/claim/public-frontier-p50",
        "canonical": {
          "value": 720,
          "unit": "expert-minutes",
          "qualifier": "ca."
        }
      },
      {
        "id": "agent-horizon/frontier-2026/public-p80",
        "label": "Offentlig frontier, 80 % horisont",
        "value": "ca. 1,5 t [50m–2t40m]",
        "detail": "Samme frontierkohort med strengere pålitelighetskrav.",
        "classification": "observation",
        "claimId": "agent-horizon/claim/public-frontier-p80",
        "canonical": {
          "value": 90,
          "unit": "expert-minutes",
          "qualifier": "ca."
        }
      },
      {
        "id": "agent-horizon/frontier-2026/reliability-gap",
        "label": "P50/P80 punktestimat",
        "value": "8×",
        "detail": "12 timer / 1,5 timer; illustrerer sensitivitet for suksesskravet.",
        "classification": "interpretation",
        "claimId": "agent-horizon/claim/reliability-gap",
        "canonical": {
          "value": 8,
          "unit": "multiple"
        }
      },
      {
        "id": "agent-horizon/th11/long-task-boundary",
        "label": "Målegrense",
        "value": ">16 t upålitelig",
        "detail": "METRs eksplisitte grense for dagens suite.",
        "classification": "gap",
        "claimId": "agent-horizon/claim/long-task-limit",
        "canonical": {
          "value": 960,
          "unit": "expert-minutes",
          "qualifier": "above"
        }
      },
      {
        "id": "agent-horizon/metr/gpt-5-4/p50",
        "label": "GPT-5.4, P50",
        "value": "341.735276 min",
        "detail": "186.581591–768.779526 expert-minutes.",
        "classification": "observation",
        "claimId": "agent-horizon/claim/gpt-5-4-p50",
        "canonical": {
          "value": 341.735276,
          "unit": "expert-minutes"
        }
      },
      {
        "id": "agent-horizon/metr/gpt-5-4/p80",
        "label": "GPT-5.4, P80",
        "value": "53.877851 min",
        "detail": "23.957027–108.679232 expert-minutes.",
        "classification": "observation",
        "claimId": "agent-horizon/claim/gpt-5-4-p80",
        "canonical": {
          "value": 53.877851,
          "unit": "expert-minutes"
        }
      }
    ],
    "limitations": [
      "Dette er primært programvare-, ML- og cybersikkerhetsoppgaver, ikke alle økonomiske oppgaver.",
      "Horisonten er logistisk modellert suksesssannsynlighet, ikke garantert autonom arbeidstid.",
      "Verdier nær og over 16 timer er svært usikre fordi suiten har få lange oppgaver."
    ]
  },
  "facts": {
    "schemaVersion": 1,
    "moduleId": "agent-horizon",
    "facts": [
      {
        "id": "agent-horizon/fact/metr/opus-4-5-p50",
        "observedAt": "2026-01-29",
        "classification": "observation",
        "label": "Claude Opus 4.5 TH1.1 P50",
        "value": "320 [170,729] minutes",
        "sourceId": "metr-time-horizon-1-1-2026",
        "sourceLocator": "Changes to Model Horizon Estimates table",
        "canonical": {
          "value": 320,
          "unit": "expert-minutes"
        }
      },
      {
        "id": "agent-horizon/fact/metr/suite",
        "observedAt": "2026-01-29",
        "classification": "observation",
        "label": "TH1.1 task suite",
        "value": "228 tasks; 31 at 8h+; 5 human baselined",
        "sourceId": "metr-time-horizon-1-1-2026",
        "sourceLocator": "Task suite description"
      },
      {
        "id": "agent-horizon/fact/metr/public-frontier-p50",
        "observedAt": "2026-03-31",
        "classification": "observation",
        "label": "Public frontier TH1.1 P50",
        "value": "about 12h [5h,61h]",
        "sourceId": "metr-frontier-risk-report-2026",
        "sourceLocator": "Table 1: public frontier Feb–Mar 2026",
        "canonical": {
          "value": 720,
          "unit": "expert-minutes",
          "qualifier": "ca.; interval 300–3660"
        }
      },
      {
        "id": "agent-horizon/fact/metr/public-frontier-p80",
        "observedAt": "2026-03-31",
        "classification": "observation",
        "label": "Public frontier TH1.1 P80",
        "value": "about 1.5h [50m,2h40m]",
        "sourceId": "metr-frontier-risk-report-2026",
        "sourceLocator": "Table 1: public frontier Feb–Mar 2026",
        "canonical": {
          "value": 90,
          "unit": "expert-minutes",
          "qualifier": "ca.; interval 50–160"
        }
      },
      {
        "id": "agent-horizon/fact/metr/limit",
        "observedAt": "2026-05-08",
        "classification": "gap",
        "label": "Long task reliability boundary",
        "value": "above 16h unreliable",
        "sourceId": "metr-time-horizons-2026",
        "sourceLocator": "Methodological Details",
        "canonical": {
          "value": 960,
          "unit": "expert-minutes",
          "qualifier": "above"
        }
      },
      {
        "id": "agent-horizon/fact/metr/gpt-5-4-p50",
        "observedAt": "2026-05-08",
        "classification": "observation",
        "label": "GPT-5.4 TH1.1 P50",
        "value": "341.735276 [186.581591,768.779526] minutes",
        "sourceId": "agent-horizon-metr-th11-results-aae31902b051",
        "sourceLocator": "results.gpt_5_4.metrics.p50_horizon_length and results.gpt_5_4.metrics.p80_horizon_length",
        "canonical": {
          "value": 341.735276,
          "unit": "expert-minutes",
          "qualifier": "interval 186.581591-768.779526"
        }
      },
      {
        "id": "agent-horizon/fact/metr/gpt-5-4-p80",
        "observedAt": "2026-05-08",
        "classification": "observation",
        "label": "GPT-5.4 TH1.1 P80",
        "value": "53.877851 [23.957027,108.679232] minutes",
        "sourceId": "agent-horizon-metr-th11-results-aae31902b051",
        "sourceLocator": "results.gpt_5_4.metrics.p50_horizon_length and results.gpt_5_4.metrics.p80_horizon_length",
        "canonical": {
          "value": 53.877851,
          "unit": "expert-minutes",
          "qualifier": "interval 23.957027-108.679232"
        }
      }
    ]
  },
  "measurements": {
    "schemaVersion": 1,
    "moduleId": "agent-horizon",
    "measurements": [
      {
        "id": "agent-horizon/th11/opus-4-5-p50",
        "factIds": [
          "agent-horizon/fact/metr/opus-4-5-p50"
        ],
        "observedAt": "2026-01-29",
        "classification": "observation",
        "label": "Claude Opus 4.5, 50 % horisont",
        "value": "320 min [170–729]",
        "detail": "P50 logistisk fit.",
        "canonical": {
          "value": 320,
          "unit": "expert-minutes"
        }
      },
      {
        "id": "agent-horizon/th11/task-suite",
        "factIds": [
          "agent-horizon/fact/metr/suite"
        ],
        "observedAt": "2026-01-29",
        "classification": "interpretation",
        "label": "TH1.1 task suite",
        "value": "228 oppgaver",
        "detail": "31 er 8+ timer; fem har menneskelig baseline.",
        "canonical": {
          "value": 228,
          "unit": "tasks"
        }
      },
      {
        "id": "agent-horizon/th11/long-task-coverage",
        "factIds": [
          "agent-horizon/fact/metr/suite"
        ],
        "observedAt": "2026-01-29",
        "classification": "interpretation",
        "label": "Langoppgavedekning",
        "value": "13,6 % / 2,2 %",
        "detail": "Andel 8t+ / andel 8t+ med menneskelig baseline."
      },
      {
        "id": "agent-horizon/frontier-2026/public-p50",
        "factIds": [
          "agent-horizon/fact/metr/public-frontier-p50"
        ],
        "observedAt": "2026-03-31",
        "classification": "observation",
        "label": "Offentlig frontier, 50 %",
        "value": "ca. 12 t [5–61]",
        "detail": "Kohortresultat, bred usikkerhet.",
        "canonical": {
          "value": 720,
          "unit": "expert-minutes",
          "qualifier": "ca."
        }
      },
      {
        "id": "agent-horizon/frontier-2026/public-p80",
        "factIds": [
          "agent-horizon/fact/metr/public-frontier-p80"
        ],
        "observedAt": "2026-03-31",
        "classification": "observation",
        "label": "Offentlig frontier, 80 %",
        "value": "ca. 1,5 t [50m–2t40m]",
        "detail": "Strengere suksesskrav.",
        "canonical": {
          "value": 90,
          "unit": "expert-minutes",
          "qualifier": "ca."
        }
      },
      {
        "id": "agent-horizon/frontier-2026/reliability-gap",
        "factIds": [
          "agent-horizon/fact/metr/public-frontier-p50",
          "agent-horizon/fact/metr/public-frontier-p80"
        ],
        "observedAt": "2026-03-31",
        "classification": "interpretation",
        "label": "P50/P80 punktestimat",
        "value": "8×",
        "detail": "720 / 90 ekspertminutter.",
        "canonical": {
          "value": 8,
          "unit": "multiple"
        }
      },
      {
        "id": "agent-horizon/th11/long-task-boundary",
        "factIds": [
          "agent-horizon/fact/metr/limit"
        ],
        "observedAt": "2026-05-08",
        "classification": "gap",
        "label": "Målegrense",
        "value": ">16 t upålitelig",
        "detail": "Kildeoppgitt begrensning.",
        "canonical": {
          "value": 960,
          "unit": "expert-minutes",
          "qualifier": "above"
        }
      },
      {
        "id": "agent-horizon/metr/gpt-5-4/p50",
        "factIds": [
          "agent-horizon/fact/metr/gpt-5-4-p50"
        ],
        "observedAt": "2026-05-08",
        "classification": "observation",
        "label": "GPT-5.4, P50",
        "value": "341.735276 min [186.581591–768.779526]",
        "detail": "TH1.1 logistic fit.",
        "canonical": {
          "value": 341.735276,
          "unit": "expert-minutes"
        }
      },
      {
        "id": "agent-horizon/metr/gpt-5-4/p80",
        "factIds": [
          "agent-horizon/fact/metr/gpt-5-4-p80"
        ],
        "observedAt": "2026-05-08",
        "classification": "observation",
        "label": "GPT-5.4, P80",
        "value": "53.877851 min [23.957027–108.679232]",
        "detail": "TH1.1 logistic fit.",
        "canonical": {
          "value": 53.877851,
          "unit": "expert-minutes"
        }
      }
    ]
  },
  "claims": {
    "schemaVersion": 1,
    "moduleId": "agent-horizon",
    "generatedAt": "2026-08-23T00:52:22.230Z",
    "claims": [
      {
        "id": "agent-horizon/claim/opus-p50",
        "claimType": "observation",
        "statement": "METR TH1.1 rapporterer 320 minutter [170–729] for Claude Opus 4.5 ved modellert 50 % suksess.",
        "asOf": "2026-01-29",
        "reviewState": "reviewed",
        "sourceIds": [
          "metr-time-horizon-1-1-2026"
        ],
        "measurementIds": [
          "agent-horizon/th11/opus-4-5-p50"
        ],
        "factIds": [
          "agent-horizon/fact/metr/opus-4-5-p50"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Spesifikk modell, suite og agentoppsett."
        ],
        "approvedAt": "2026-08-13T09:38:37.386Z"
      },
      {
        "id": "agent-horizon/claim/public-frontier-p50",
        "claimType": "observation",
        "statement": "METRs vurdering for februar–mars 2026 setter offentlig frontier til omtrent 12 timer [5–61] ved 50 % suksess.",
        "asOf": "2026-03-31",
        "reviewState": "reviewed",
        "sourceIds": [
          "metr-frontier-risk-report-2026"
        ],
        "measurementIds": [
          "agent-horizon/frontier-2026/public-p50"
        ],
        "factIds": [
          "agent-horizon/fact/metr/public-frontier-p50"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Kohortestimat med bredt intervall og suite nær metning."
        ],
        "approvedAt": "2026-08-13T09:38:37.386Z"
      },
      {
        "id": "agent-horizon/claim/public-frontier-p80",
        "claimType": "observation",
        "statement": "Samme METR-vurdering setter offentlig frontier til omtrent 1,5 timer [50 minutter–2 timer 40 minutter] ved 80 % suksess.",
        "asOf": "2026-03-31",
        "reviewState": "reviewed",
        "sourceIds": [
          "metr-frontier-risk-report-2026"
        ],
        "measurementIds": [
          "agent-horizon/frontier-2026/public-p80"
        ],
        "factIds": [
          "agent-horizon/fact/metr/public-frontier-p80"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Kohortestimat, ikke navngitt modellrangering."
        ],
        "approvedAt": "2026-08-13T09:38:37.386Z"
      },
      {
        "id": "agent-horizon/claim/reliability-gap",
        "claimType": "derivation",
        "statement": "Punktestimatet for offentlig frontier er åtte ganger lengre ved 50 % enn ved 80 % suksess, noe som viser sterk følsomhet for pålitelighetskravet.",
        "asOf": "2026-03-31",
        "reviewState": "reviewed",
        "sourceIds": [
          "metr-frontier-risk-report-2026"
        ],
        "measurementIds": [
          "agent-horizon/frontier-2026/public-p50",
          "agent-horizon/frontier-2026/public-p80",
          "agent-horizon/frontier-2026/reliability-gap"
        ],
        "factIds": [
          "agent-horizon/fact/metr/public-frontier-p50",
          "agent-horizon/fact/metr/public-frontier-p80"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": {
          "formula": "12 hours / 1.5 hours = 8",
          "inputs": [
            "agent-horizon/fact/metr/public-frontier-p50",
            "agent-horizon/fact/metr/public-frontier-p80"
          ],
          "outputUnit": "multiple"
        },
        "limitations": [
          "Forholdet bruker avrundede punktestimater; intervallene er brede."
        ],
        "approvedAt": "2026-08-13T09:38:37.386Z"
      },
      {
        "id": "agent-horizon/claim/suite-coverage",
        "claimType": "method",
        "statement": "TH1.1 har 228 oppgaver, hvorav 31 er estimert til å ta minst åtte timer og fem av disse har menneskelig baseline.",
        "asOf": "2026-01-29",
        "reviewState": "reviewed",
        "sourceIds": [
          "metr-time-horizon-1-1-2026"
        ],
        "measurementIds": [
          "agent-horizon/th11/task-suite"
        ],
        "factIds": [
          "agent-horizon/fact/metr/suite"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Ikke representativ arbeidsmarkedsfordeling."
        ],
        "approvedAt": "2026-08-13T09:38:37.386Z"
      },
      {
        "id": "agent-horizon/claim/long-task-coverage",
        "claimType": "derivation",
        "statement": "Bare 13,6 % av TH1.1-oppgavene er åtte timer eller lengre, og 2,2 % av hele suiten er slike oppgaver med menneskelig baseline.",
        "asOf": "2026-01-29",
        "reviewState": "reviewed",
        "sourceIds": [
          "metr-time-horizon-1-1-2026"
        ],
        "measurementIds": [
          "agent-horizon/th11/long-task-coverage"
        ],
        "factIds": [
          "agent-horizon/fact/metr/suite"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": {
          "formula": "31 / 228 = 13.6%; 5 / 228 = 2.2%",
          "inputs": [
            "agent-horizon/fact/metr/suite"
          ],
          "outputUnit": "% of tasks"
        },
        "limitations": [
          "Oppgaveantall sier ikke noe om representativitet eller kvalitet."
        ],
        "approvedAt": "2026-08-13T09:38:37.386Z"
      },
      {
        "id": "agent-horizon/claim/long-task-limit",
        "claimType": "limitation",
        "statement": "METR markerer anslag over 16 timer som upålitelige med dagens task suite.",
        "asOf": "2026-05-08",
        "reviewState": "reviewed",
        "sourceIds": [
          "metr-time-horizons-2026",
          "metr-frontier-risk-report-2026"
        ],
        "measurementIds": [
          "agent-horizon/th11/long-task-boundary"
        ],
        "factIds": [
          "agent-horizon/fact/metr/limit"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Ikke et utsagn om faktisk maksimal autonomi."
        ],
        "approvedAt": "2026-08-13T09:38:37.386Z"
      },
      {
        "id": "agent-horizon/claim/evidence-state",
        "claimType": "method",
        "statement": "Agent Horizon viser 4 kompatible P50/P80-estimater på tvers av 2 offentlige modellkohorter i samme TH1.1-kontrakt. De er samtidige modell- og reliabilitetsobservasjoner, ikke en tidsserie; suitedekning og 16-timersgrensen begrenser overførbarheten.",
        "asOf": "2026-08-21",
        "reviewState": "published",
        "sourceIds": [
          "metr-time-horizon-1-1-2026",
          "metr-time-horizons-2026",
          "metr-frontier-risk-report-2026",
          "agent-horizon-metr-th11-results-aae31902b051"
        ],
        "measurementIds": [
          "agent-horizon/frontier-2026/public-p50",
          "agent-horizon/frontier-2026/public-p80",
          "agent-horizon/th11/task-suite",
          "agent-horizon/th11/long-task-boundary",
          "agent-horizon/metr/gpt-5-4/p50",
          "agent-horizon/metr/gpt-5-4/p80"
        ],
        "factIds": [
          "agent-horizon/fact/metr/public-frontier-p50",
          "agent-horizon/fact/metr/public-frontier-p80",
          "agent-horizon/fact/metr/suite",
          "agent-horizon/fact/metr/limit",
          "agent-horizon/fact/metr/gpt-5-4-p50",
          "agent-horizon/fact/metr/gpt-5-4-p80"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "P50 og P80 er reliabilitetsnivåer, ikke separate tidspunkt.",
          "Suiten er avgrenset til programvare-, ML- og cybersikkerhetsoppgaver."
        ],
        "supersedesRevisionId": "agent-horizon/claim/evidence-state@d58612a37f21",
        "approvedAt": "2026-08-23T00:52:22.230Z"
      },
      {
        "id": "agent-horizon/claim/gpt-5-4-p50",
        "claimType": "observation",
        "statement": "METR TH1.1 reports 341.735276 expert-minutes [186.581591–768.779526] for GPT-5.4 at 50% success.",
        "asOf": "2026-05-08",
        "reviewState": "published",
        "sourceIds": [
          "agent-horizon-metr-th11-results-aae31902b051"
        ],
        "measurementIds": [
          "agent-horizon/metr/gpt-5-4/p50"
        ],
        "factIds": [
          "agent-horizon/fact/metr/gpt-5-4-p50"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Model-, suite- and setup-specific estimate."
        ],
        "approvedAt": "2026-08-23T00:52:22.230Z"
      },
      {
        "id": "agent-horizon/claim/gpt-5-4-p80",
        "claimType": "observation",
        "statement": "METR TH1.1 reports 53.877851 expert-minutes [23.957027–108.679232] for GPT-5.4 at 80% success.",
        "asOf": "2026-05-08",
        "reviewState": "published",
        "sourceIds": [
          "agent-horizon-metr-th11-results-aae31902b051"
        ],
        "measurementIds": [
          "agent-horizon/metr/gpt-5-4/p80"
        ],
        "factIds": [
          "agent-horizon/fact/metr/gpt-5-4-p80"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Model-, suite- and setup-specific estimate."
        ],
        "approvedAt": "2026-08-23T00:52:22.230Z"
      }
    ],
    "claimHistory": [
      {
        "id": "agent-horizon/claim/evidence-state",
        "claimType": "method",
        "statement": "Agent Horizon viser to kompatible reliabilitetsnivåer, P50 og P80, for samme offentlige frontierkohort. De er samtidige valg av suksesskrav og utgjør ikke en tidsserie; suitedekning og 16-timersgrensen begrenser overførbarheten.",
        "asOf": "2026-08-21",
        "reviewState": "reviewed",
        "sourceIds": [
          "metr-time-horizon-1-1-2026",
          "metr-time-horizons-2026",
          "metr-frontier-risk-report-2026"
        ],
        "measurementIds": [
          "agent-horizon/frontier-2026/public-p50",
          "agent-horizon/frontier-2026/public-p80",
          "agent-horizon/th11/task-suite",
          "agent-horizon/th11/long-task-boundary"
        ],
        "factIds": [],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/agent-horizon/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "P50 og P80 er reliabilitetsnivåer, ikke separate tidspunkt.",
          "Suiten er avgrenset til programvare-, ML- og cybersikkerhetsoppgaver."
        ],
        "revisionId": "agent-horizon/claim/evidence-state@d58612a37f21",
        "supersededAt": "2026-08-23T00:52:22.230Z",
        "supersededByProposalId": "proposal.agent-horizon.aae31902b051.gpt-5-4-th11"
      }
    ]
  },
  "drilldowns": {
    "schemaVersion": 1,
    "moduleId": "agent-horizon",
    "views": [
      {
        "schemaVersion": 1,
        "id": "reliability",
        "moduleId": "agent-horizon",
        "title": "Horisont er et pålitelighetsvalg",
        "shortTitle": "Agenter",
        "description": "Den samme frontierkohorten ser radikalt forskjellig ut ved 50 og 80 prosent suksess. Suitedekningen setter en hard fortolkningsgrense.",
        "decisionQuestion": "Hvor lange oppgaver kan agenter fullføre med den påliteligheten brukssituasjonen faktisk krever?",
        "decisionUse": "Velg pålitelighetsnivå før du tolker autonomi, og stopp når benchmarkens lange oppgaver ikke lenger bærer estimatet.",
        "asOf": "2026-05-08",
        "sourceIds": [
          "metr-time-horizon-1-1-2026",
          "metr-time-horizons-2026",
          "metr-frontier-risk-report-2026",
          "agent-horizon-metr-th11-results-aae31902b051"
        ],
        "headline": {
          "label": "Offentlig frontier ved 80 % suksess",
          "value": "ca. 1,5 t",
          "detail": "Det strengere målet er ofte mer beslutningsrelevant enn P50.",
          "classification": "observation",
          "claimId": "agent-horizon/claim/public-frontier-p80"
        },
        "visualization": {
          "id": "agent-reliability-intervals",
          "type": "agent-log-intervals",
          "eyebrow": "HORISONTPLOTT / USIKKERHET + RELIABILITET",
          "title": "Pålitelighet komprimerer den brukbare horisonten",
          "description": "Punktet er modellert horisont; streken er kildeintervallet. Logaritmisk x-akse gjør 50 minutter og 61 timer lesbare i samme plot.",
          "measurementIds": [
            "agent-horizon/th11/opus-4-5-p50",
            "agent-horizon/frontier-2026/public-p50",
            "agent-horizon/frontier-2026/public-p80",
            "agent-horizon/th11/long-task-boundary",
            "agent-horizon/th11/task-suite",
            "agent-horizon/th11/long-task-coverage",
            "agent-horizon/metr/gpt-5-4/p50",
            "agent-horizon/metr/gpt-5-4/p80"
          ],
          "claimIds": [
            "agent-horizon/claim/opus-p50",
            "agent-horizon/claim/public-frontier-p50",
            "agent-horizon/claim/public-frontier-p80",
            "agent-horizon/claim/long-task-limit",
            "agent-horizon/claim/suite-coverage",
            "agent-horizon/claim/long-task-coverage",
            "agent-horizon/claim/reliability-gap",
            "agent-horizon/claim/gpt-5-4-p50",
            "agent-horizon/claim/gpt-5-4-p80"
          ],
          "unit": "ekspertminutter, log-skala",
          "scale": {
            "min": 30,
            "max": 4000,
            "ticks": [
              {
                "value": 30,
                "label": "30m"
              },
              {
                "value": 60,
                "label": "1t"
              },
              {
                "value": 120,
                "label": "2t"
              },
              {
                "value": 240,
                "label": "4t"
              },
              {
                "value": 480,
                "label": "8t"
              },
              {
                "value": 960,
                "label": "16t"
              },
              {
                "value": 1920,
                "label": "32t"
              },
              {
                "value": 3840,
                "label": "64t"
              }
            ]
          },
          "ranges": [
            {
              "id": "opus-4-5",
              "label": "Opus 4.5 / P50",
              "displayValue": "320 min",
              "low": 170,
              "median": 320,
              "high": 729,
              "kind": "context",
              "measurementId": "agent-horizon/th11/opus-4-5-p50",
              "claimId": "agent-horizon/claim/opus-p50"
            },
            {
              "id": "frontier-p50",
              "label": "Frontier / P50",
              "displayValue": "ca. 12 t",
              "low": 300,
              "median": 720,
              "high": 3660,
              "kind": "frontier",
              "measurementId": "agent-horizon/frontier-2026/public-p50",
              "claimId": "agent-horizon/claim/public-frontier-p50"
            },
            {
              "id": "frontier-p80",
              "label": "Frontier / P80",
              "displayValue": "ca. 1,5 t",
              "low": 50,
              "median": 90,
              "high": 160,
              "kind": "frontier",
              "measurementId": "agent-horizon/frontier-2026/public-p80",
              "claimId": "agent-horizon/claim/public-frontier-p80"
            },
            {
              "id": "gpt-5-4-p50",
              "label": "GPT-5.4 / P50",
              "displayValue": "341.735276 min",
              "low": 186.581591,
              "median": 341.735276,
              "high": 768.779526,
              "kind": "frontier",
              "measurementId": "agent-horizon/metr/gpt-5-4/p50",
              "claimId": "agent-horizon/claim/gpt-5-4-p50"
            },
            {
              "id": "gpt-5-4-p80",
              "label": "GPT-5.4 / P80",
              "displayValue": "53.877851 min",
              "low": 23.957027,
              "median": 53.877851,
              "high": 108.679232,
              "kind": "frontier",
              "measurementId": "agent-horizon/metr/gpt-5-4/p80",
              "claimId": "agent-horizon/claim/gpt-5-4-p80"
            }
          ],
          "reliabilityBoundary": {
            "value": 960,
            "label": "16t · OVER HER UPÅLITELIG",
            "measurementId": "agent-horizon/th11/long-task-boundary",
            "claimId": "agent-horizon/claim/long-task-limit"
          },
          "coverage": {
            "total": 228,
            "long": 31,
            "longWithBaseline": 5,
            "measurementIds": [
              "agent-horizon/th11/task-suite",
              "agent-horizon/th11/long-task-coverage"
            ],
            "claimIds": [
              "agent-horizon/claim/suite-coverage",
              "agent-horizon/claim/long-task-coverage"
            ]
          }
        },
        "sections": [
          {
            "id": "p50",
            "label": "50 % SUKSESS",
            "question": "Hvor langt ved myntkast-pålitelighet?",
            "conclusion": "Offentlig frontier ligger rundt 12 timer, men intervallet er 5–61 timer og går inn i metningsområdet.",
            "status": "signal",
            "measurementIds": [
              "agent-horizon/frontier-2026/public-p50"
            ],
            "claimIds": [
              "agent-horizon/claim/public-frontier-p50"
            ]
          },
          {
            "id": "p80",
            "label": "80 % SUKSESS",
            "question": "Hvor langt ved høyere pålitelighet?",
            "conclusion": "Punktestimatet faller til ca. 1,5 timer med intervall 50 minutter–2 timer 40 minutter.",
            "status": "signal",
            "measurementIds": [
              "agent-horizon/frontier-2026/public-p80"
            ],
            "claimIds": [
              "agent-horizon/claim/public-frontier-p80"
            ]
          },
          {
            "id": "reliability-sensitivity",
            "label": "FØLSOMHET",
            "question": "Hvor mye betyr suksesskravet?",
            "conclusion": "De avrundede P50- og P80-punktestimatene skiller åtte ganger.",
            "status": "constraint",
            "measurementIds": [
              "agent-horizon/frontier-2026/reliability-gap"
            ],
            "claimIds": [
              "agent-horizon/claim/reliability-gap"
            ]
          },
          {
            "id": "suite-coverage",
            "label": "SUITEDEKNING",
            "question": "Hvor mye langt arbeid finnes i målegrunnlaget?",
            "conclusion": "31 av 228 oppgaver er 8+ timer; bare fem av hele suiten er lange oppgaver med menneskelig baseline.",
            "status": "constraint",
            "measurementIds": [
              "agent-horizon/th11/task-suite",
              "agent-horizon/th11/long-task-coverage"
            ],
            "claimIds": [
              "agent-horizon/claim/suite-coverage",
              "agent-horizon/claim/long-task-coverage"
            ]
          },
          {
            "id": "measurement-ceiling",
            "label": "MÅLEGAP",
            "question": "Når bør tallet ikke brukes?",
            "conclusion": "METR markerer horisonter over 16 timer som upålitelige med dagens suite.",
            "status": "gap",
            "measurementIds": [
              "agent-horizon/th11/long-task-boundary"
            ],
            "claimIds": [
              "agent-horizon/claim/long-task-limit"
            ]
          }
        ],
        "comparisons": [
          {
            "id": "agent-reliability-comparison",
            "label": "PÅLITELIGHETSFALL",
            "from": "ca. 12 t @ 50 %",
            "to": "ca. 1,5 t @ 80 %",
            "change": "8× kortere",
            "detail": "Punktestimater for samme offentlige frontierkohort.",
            "claimId": "agent-horizon/claim/reliability-gap"
          },
          {
            "id": "agent-model-to-frontier",
            "label": "MÅLEUTVIKLING",
            "from": "Opus 4.5: 320 min",
            "to": "Frontierkohort: ca. 720 min",
            "change": "ulike vurderinger",
            "detail": "Vises som kontekst, ikke som ren modelltrend.",
            "claimId": "agent-horizon/claim/opus-p50"
          }
        ],
        "limitations": [
          "Frontierkohorten er ikke en navngitt modellrangering.",
          "Intervallene er brede og metning påvirker P50.",
          "Programvareoppgaver kan ikke generaliseres direkte til vanlig kunnskapsarbeid."
        ]
      }
    ]
  },
  "sources": [
    {
      "id": "metr-time-horizon-1-1-2026",
      "title": "Time Horizon 1.1",
      "publisher": "METR",
      "sourceUrl": "https://metr.org/blog/2026-1-29-time-horizon-1-1/",
      "documentationUrl": "https://metr.org/time-horizons/",
      "license": "Cited public research release; analysis repository license applies to its code/data",
      "retrievedAt": "2026-08-12",
      "publishedAt": "2026-01-29",
      "dataAsOf": "2026-01-29",
      "sha256": null,
      "localPath": null,
      "role": "primary-measurement",
      "notes": "Public, task-suite-specific agent horizon estimates; this module records only stated values, not a copied raw dataset."
    },
    {
      "id": "metr-time-horizons-2026",
      "title": "Task-Completion Time Horizons of Frontier AI Models",
      "publisher": "METR",
      "sourceUrl": "https://metr.org/time-horizons/",
      "documentationUrl": "https://github.com/METR/eval-analysis-public",
      "license": "Cited public web source; repository license applies separately",
      "retrievedAt": "2026-08-12",
      "publishedAt": "2026-05-08",
      "dataAsOf": "2026-05-08",
      "sha256": null,
      "localPath": null,
      "role": "measurement-methodology",
      "notes": "Defines time horizon and explicitly marks estimates above 16 hours unreliable in the current suite."
    },
    {
      "id": "metr-frontier-risk-report-2026",
      "title": "Frontier Risk Report: February to March 2026",
      "publisher": "METR",
      "sourceUrl": "https://metr.org/blog/2026-05-19-frontier-risk-report/",
      "documentationUrl": "https://metr.org/risk-report-feb-mar-2026.pdf",
      "license": "Cited public research report",
      "retrievedAt": "2026-08-12",
      "publishedAt": "2026-05-19",
      "dataAsOf": "2026-03-31",
      "sha256": null,
      "localPath": null,
      "role": "primary-measurement",
      "notes": "Cohort-level public-frontier TH1.1 results at 50% and 80% reliability; saturation and wide intervals are material limitations."
    },
    {
      "id": "agent-horizon-metr-th11-results-aae31902b051",
      "title": "METR — Time Horizon 1.1 benchmark results",
      "publisher": "METR",
      "sourceUrl": "https://metr.org/assets/benchmark_results_1_1.yaml",
      "documentationUrl": "https://metr.org/assets/benchmark_results_1_1.yaml",
      "license": "citation-only-review-required",
      "retrievedAt": "2026-08-23",
      "publishedAt": "2026-05-08",
      "dataAsOf": "2026-05-08",
      "sha256": "aae31902b0519a4da73e16643915e5e8aca13cd3315c3aac893ce3d6dfe92ad9",
      "localPath": null,
      "role": "reviewed-research-source",
      "notes": "METR's official machine-readable TH1.1 result artifact reports GPT-5.4 at 341.735276 expert-minutes P50 [186.581591–768.779526] and 53.877851 expert-minutes P80 [23.957027–108.679232]. Both estimates remain below METR's 16-hour reliability boundary and fit the accepted Agent Horizon adapter without a methodology change.",
      "admission": {
        "candidateId": "research.agent-horizon.2026-08-23.gpt-5-4-th11",
        "admittedAt": "2026-08-23T00:52:22.230Z",
        "contentFingerprint": "aae31902b0519a4da73e16643915e5e8aca13cd3315c3aac893ce3d6dfe92ad9",
        "rawSha256": "aae31902b0519a4da73e16643915e5e8aca13cd3315c3aac893ce3d6dfe92ad9",
        "mediaType": "application/x-yaml",
        "finalUrl": "https://metr.org/assets/benchmark_results_1_1.yaml",
        "archiveException": {
          "rationale": "Official citation-only METR result artifact; retain its URL, normalized fingerprint, raw checksum and exact source locator without redistributing source bytes.",
          "acceptedBy": "automation/tier1-verifier/agent-horizon-v1",
          "acceptedAt": "2026-08-23T00:52:22.230Z"
        }
      }
    }
  ]
}