{
  "schemaVersion": 1,
  "moduleId": "research-automation",
  "generatedAt": "2026-08-13T09:42:11.000Z",
  "evidenceState": {
    "claimId": "research-automation/claim/evidence-state",
    "lastSourceCheckAt": "2026-08-21",
    "maturity": {
      "level": "preliminary",
      "label": "TIDLIG",
      "rationale": "Tre fagfellevurderbare benchmarkfamilier belyser ulike deler av FoU-løkken, men kan ikke aggregeres eller leses som én tidsserie."
    },
    "cadence": "editorial",
    "compatibleObservationCount": 0,
    "compatibleObservationLabel": "longitudinelle punkter innen samme benchmarkversjon",
    "trend": {
      "status": "not-established",
      "label": "IKKE ETABLERT",
      "rationale": "RExBench, MLR-Bench og MLRC-Bench har forskjellige oppgaver, evaluatorer og nevnere."
    },
    "measurementGaps": [
      {
        "label": "samlet automatiseringsandel",
        "claimId": "research-automation/claim/no-aggregate"
      },
      {
        "label": "uavhengig validering og reproduksjon",
        "claimId": "research-automation/claim/validation-reproduction-gap"
      }
    ]
  },
  "snapshot": {
    "schemaVersion": 1,
    "moduleId": "research-automation",
    "title": "AI Research Automation",
    "eyebrow": "KAPABILITET / AI-FoU",
    "asOf": "2025-06-27",
    "retrievedAt": "2026-08-13",
    "generatedAt": "2026-08-13T09:42:11.000Z",
    "sourceIds": [
      "rexbench-2025",
      "mlr-bench-2025",
      "mlrc-bench-2025",
      "research-automation-rexbench-paper-06e0d5161012"
    ],
    "status": "incomplete",
    "evidenceState": {
      "claimId": "research-automation/claim/evidence-state",
      "lastSourceCheckAt": "2026-08-21",
      "maturity": {
        "level": "preliminary",
        "label": "TIDLIG",
        "rationale": "Tre fagfellevurderbare benchmarkfamilier belyser ulike deler av FoU-løkken, men kan ikke aggregeres eller leses som én tidsserie."
      },
      "cadence": "editorial",
      "compatibleObservationCount": 0,
      "compatibleObservationLabel": "longitudinelle punkter innen samme benchmarkversjon",
      "trend": {
        "status": "not-established",
        "label": "IKKE ETABLERT",
        "rationale": "RExBench, MLR-Bench og MLRC-Bench har forskjellige oppgaver, evaluatorer og nevnere."
      },
      "measurementGaps": [
        {
          "label": "samlet automatiseringsandel",
          "claimId": "research-automation/claim/no-aggregate"
        },
        {
          "label": "uavhengig validering og reproduksjon",
          "claimId": "research-automation/claim/validation-reproduction-gap"
        }
      ]
    },
    "decisionQuestion": "Hvilke deler av AI-forskning er demonstrert i målinger, og hvor stopper pålitelig, objektivt evaluerbar autonomi?",
    "summary": "Tre separate benchmarklinser; eksperimentell validitet er den tydeligste flaskehalsen",
    "valueStatement": "Se hvilke FoU-steg agenter faktisk demonstrerer—og hvor resultatene fortsatt svikter objektiv eller eksperimentell kontroll.",
    "valueClaimId": "research-automation/claim/experimental-bottleneck",
    "analysisRoute": "/tracker/research-automation/benchmarks",
    "measurements": [
      {
        "id": "research-automation/rexbench/best-with-hints",
        "label": "RExBench: forskningsutvidelser",
        "value": "under 44 %",
        "detail": "Beste testede resultat selv med menneskeskrevne hints; 12 oppgaver.",
        "classification": "observation",
        "claimId": "research-automation/claim/rexbench-capability",
        "canonical": {
          "value": 44,
          "unit": "%",
          "qualifier": "under"
        }
      },
      {
        "id": "research-automation/mlrc/gap-closed",
        "label": "MLRC-Bench: gap mot toppmenneske",
        "value": "9,3 %",
        "detail": "Beste testede agent på syv objektivt evaluerte konkurranseoppgaver.",
        "classification": "observation",
        "claimId": "research-automation/claim/mlrc-capability",
        "canonical": {
          "value": 9.3,
          "unit": "% of human-performance gap"
        }
      },
      {
        "id": "research-automation/mlr/invalid-experiments",
        "label": "MLR-Bench: ugyldige eksperimentresultater",
        "value": "ofte 80 %",
        "detail": "Rapportert for ett coding-agentoppsett; ikke universell feilrate.",
        "classification": "observation",
        "claimId": "research-automation/claim/mlr-reliability",
        "canonical": {
          "value": 80,
          "unit": "%",
          "qualifier": "eksempelvis"
        }
      },
      {
        "id": "research-automation/mlr/workflow-scope",
        "label": "MLR-Bench: ende-til-ende-dekning",
        "value": "201 oppgaver / 4 steg",
        "detail": "Idé, forslag, eksperiment og paperskriving; benchmarkdekning, ikke automatiseringsandel.",
        "classification": "interpretation",
        "claimId": "research-automation/claim/mlr-scope"
      }
    ],
    "limitations": [
      "Benchmarkene har ulike oppgaver, agenter og evalueringsmetoder og skal ikke aggregeres.",
      "Åpne benchmarker observerer ikke intern laboratoriepraksis.",
      "Modulen har ingen legitim tidsserie før benchmarkversjoner kan sammenlignes på samme kontrakt."
    ]
  },
  "facts": {
    "schemaVersion": 1,
    "moduleId": "research-automation",
    "facts": [
      {
        "id": "research-automation/fact/rexbench/best-with-hints",
        "observedAt": "2025-06-27",
        "classification": "observation",
        "label": "Best RExBench performance with human-written hints",
        "value": "below 44%",
        "sourceId": "research-automation-rexbench-paper-06e0d5161012",
        "sourceLocator": "v3 abstract revised 2026-04-21",
        "canonical": {
          "value": 44,
          "unit": "%",
          "qualifier": "under"
        }
      },
      {
        "id": "research-automation/fact/rexbench/tasks",
        "observedAt": "2025-06-27",
        "classification": "observation",
        "label": "RExBench realistic research extension tasks",
        "value": "12",
        "sourceId": "rexbench-2025",
        "sourceLocator": "Abstract",
        "canonical": {
          "value": 12,
          "unit": "tasks"
        }
      },
      {
        "id": "research-automation/fact/mlr/tasks",
        "observedAt": "2025-05-26",
        "classification": "observation",
        "label": "MLR-Bench open-ended ML research tasks",
        "value": "201",
        "sourceId": "mlr-bench-2025",
        "sourceLocator": "Abstract",
        "canonical": {
          "value": 201,
          "unit": "tasks"
        }
      },
      {
        "id": "research-automation/fact/mlr/stages",
        "observedAt": "2025-05-26",
        "classification": "observation",
        "label": "MLR-Agent research workflow stages",
        "value": "4: idea, proposal, experimentation, paper writing",
        "sourceId": "mlr-bench-2025",
        "sourceLocator": "Abstract and workflow description",
        "canonical": {
          "value": 4,
          "unit": "workflow stages"
        }
      },
      {
        "id": "research-automation/fact/mlr/invalid-results",
        "observedAt": "2025-05-26",
        "classification": "observation",
        "label": "Coding-agent fabricated or invalidated experiment results",
        "value": "frequently, e.g. 80% of cases",
        "sourceId": "mlr-bench-2025",
        "sourceLocator": "Abstract",
        "canonical": {
          "value": 80,
          "unit": "%",
          "qualifier": "example for tested coding agent"
        }
      },
      {
        "id": "research-automation/fact/mlrc/tasks",
        "observedAt": "2025-04-13",
        "classification": "observation",
        "label": "MLRC-Bench competition tasks",
        "value": "7",
        "sourceId": "mlrc-bench-2025",
        "sourceLocator": "Abstract",
        "canonical": {
          "value": 7,
          "unit": "tasks"
        }
      },
      {
        "id": "research-automation/fact/mlrc/gap-closed",
        "observedAt": "2025-04-13",
        "classification": "observation",
        "label": "Best agent share of baseline-to-top-human gap closed",
        "value": "9.3%",
        "sourceId": "mlrc-bench-2025",
        "sourceLocator": "Abstract",
        "canonical": {
          "value": 9.3,
          "unit": "% of human-performance gap"
        }
      },
      {
        "id": "research-automation/fact/coverage/no-aggregate",
        "observedAt": "2026-08-12",
        "classification": "gap",
        "label": "Cross-benchmark aggregate automation score",
        "value": "not published",
        "sourceId": "rexbench-2025",
        "sourceLocator": "Module methodology; intentional non-derivation"
      },
      {
        "id": "research-automation/fact/coverage/validation-reproduction-gap",
        "observedAt": "2026-08-22",
        "classification": "gap",
        "label": "Independent validation and reproduction coverage",
        "value": "not established by current admitted benchmarks",
        "sourceId": "mlr-bench-2025",
        "sourceLocator": "Module coverage audit against the published four-stage workflow"
      }
    ]
  },
  "measurements": {
    "schemaVersion": 1,
    "moduleId": "research-automation",
    "measurements": [
      {
        "id": "research-automation/rexbench/best-with-hints",
        "factIds": [
          "research-automation/fact/rexbench/best-with-hints"
        ],
        "observedAt": "2025-06-27",
        "classification": "observation",
        "label": "RExBench: beste resultat med hints",
        "value": "under 44 %",
        "detail": "Forskningsutvidelser; ikke generell FoU-rate.",
        "canonical": {
          "value": 44,
          "unit": "%",
          "qualifier": "under"
        }
      },
      {
        "id": "research-automation/rexbench/task-count",
        "factIds": [
          "research-automation/fact/rexbench/tasks"
        ],
        "observedAt": "2025-06-27",
        "classification": "interpretation",
        "label": "RExBench-dekning",
        "value": "12 oppgaver",
        "detail": "Avgrenset benchmark.",
        "canonical": {
          "value": 12,
          "unit": "tasks"
        }
      },
      {
        "id": "research-automation/mlr/invalid-experiments",
        "factIds": [
          "research-automation/fact/mlr/invalid-results"
        ],
        "observedAt": "2025-05-26",
        "classification": "observation",
        "label": "MLR-Bench: eksperimentfeil",
        "value": "ofte 80 %",
        "detail": "Eksempeltall fra særskilt coding-agentoppsett.",
        "canonical": {
          "value": 80,
          "unit": "%",
          "qualifier": "example"
        }
      },
      {
        "id": "research-automation/mlr/workflow-scope",
        "factIds": [
          "research-automation/fact/mlr/tasks",
          "research-automation/fact/mlr/stages"
        ],
        "observedAt": "2025-05-26",
        "classification": "interpretation",
        "label": "MLR-Bench-dekning",
        "value": "201 oppgaver / 4 steg",
        "detail": "Ende-til-ende benchmarkdesign; ikke én sammenlignbar score."
      },
      {
        "id": "research-automation/mlrc/gap-closed",
        "factIds": [
          "research-automation/fact/mlrc/gap-closed"
        ],
        "observedAt": "2025-04-13",
        "classification": "observation",
        "label": "MLRC-Bench: gap lukket",
        "value": "9,3 %",
        "detail": "Objektiv resultatmåling mot baseline og toppmenneske.",
        "canonical": {
          "value": 9.3,
          "unit": "% of human-performance gap"
        }
      },
      {
        "id": "research-automation/mlrc/task-count",
        "factIds": [
          "research-automation/fact/mlrc/tasks"
        ],
        "observedAt": "2025-04-13",
        "classification": "interpretation",
        "label": "MLRC-Bench-dekning",
        "value": "7 oppgaver",
        "detail": "Dynamiske ML-konkurranseoppgaver.",
        "canonical": {
          "value": 7,
          "unit": "tasks"
        }
      },
      {
        "id": "research-automation/workflow/coverage",
        "factIds": [
          "research-automation/fact/coverage/no-aggregate"
        ],
        "observedAt": "2026-08-12",
        "classification": "gap",
        "label": "Samlet AI-FoU-automatisering",
        "value": "ikke målt",
        "detail": "Bevisst tomt aggregat."
      },
      {
        "id": "research-automation/workflow/validation-reproduction-gap",
        "factIds": [
          "research-automation/fact/coverage/validation-reproduction-gap"
        ],
        "observedAt": "2026-08-22",
        "classification": "gap",
        "label": "Uavhengig validering og reproduksjon",
        "value": "ikke etablert",
        "detail": "Ingen nåværende benchmarkbane dokumenterer en full uavhengig validerings- og reproduksjonssløyfe."
      }
    ]
  },
  "claims": {
    "schemaVersion": 1,
    "moduleId": "research-automation",
    "generatedAt": "2026-08-13T09:42:29.478Z",
    "claims": [
      {
        "id": "research-automation/claim/rexbench-capability",
        "claimType": "observation",
        "statement": "RExBench v3 rapporterer at beste resultat, selv med menneskeskrevne hints, er under 44 % på realistiske forskningsutvidelser.",
        "asOf": "2025-06-27",
        "reviewState": "published",
        "sourceIds": [
          "research-automation-rexbench-paper-06e0d5161012"
        ],
        "measurementIds": [
          "research-automation/rexbench/best-with-hints"
        ],
        "factIds": [
          "research-automation/fact/rexbench/best-with-hints"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Avhengig av oppsett og hint."
        ],
        "supersedesRevisionId": "research-automation/claim/rexbench-capability@7b85e6ff0b18",
        "approvedAt": "2026-08-13T09:42:29.478Z"
      },
      {
        "id": "research-automation/claim/rexbench-scope",
        "claimType": "method",
        "statement": "RExBench inneholder 12 forskningsimplementasjonsoppgaver med automatisk evaluering.",
        "asOf": "2025-06-27",
        "reviewState": "reviewed",
        "sourceIds": [
          "rexbench-2025"
        ],
        "measurementIds": [
          "research-automation/rexbench/task-count"
        ],
        "factIds": [
          "research-automation/fact/rexbench/tasks"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Ikke full forskningslivssyklus."
        ],
        "approvedAt": "2026-08-13T09:38:37.210Z"
      },
      {
        "id": "research-automation/claim/mlr-reliability",
        "claimType": "observation",
        "statement": "MLR-Bench beskriver hyppige, eksempelvis 80 %, fabrikkerte eller invaliderte eksperimentresultater for det testede coding-agentoppsettet.",
        "asOf": "2025-05-26",
        "reviewState": "reviewed",
        "sourceIds": [
          "mlr-bench-2025"
        ],
        "measurementIds": [
          "research-automation/mlr/invalid-experiments"
        ],
        "factIds": [
          "research-automation/fact/mlr/invalid-results"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Ikke en universell feilrate."
        ],
        "approvedAt": "2026-08-13T09:38:37.210Z"
      },
      {
        "id": "research-automation/claim/mlr-scope",
        "claimType": "method",
        "statement": "MLR-Bench dekker 201 oppgaver og en firestegs arbeidsflyt fra idé til paperskriving.",
        "asOf": "2025-05-26",
        "reviewState": "reviewed",
        "sourceIds": [
          "mlr-bench-2025"
        ],
        "measurementIds": [
          "research-automation/mlr/workflow-scope"
        ],
        "factIds": [
          "research-automation/fact/mlr/tasks",
          "research-automation/fact/mlr/stages"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Stadiene bruker ulike modeller og evalueringsmetoder."
        ],
        "approvedAt": "2026-08-13T09:38:37.210Z"
      },
      {
        "id": "research-automation/claim/mlrc-capability",
        "claimType": "observation",
        "statement": "I MLRC-Bench lukket beste testede agent 9,3 % av gapet mellom baseline og toppmenneskelig resultat.",
        "asOf": "2025-04-13",
        "reviewState": "reviewed",
        "sourceIds": [
          "mlrc-bench-2025"
        ],
        "measurementIds": [
          "research-automation/mlrc/gap-closed"
        ],
        "factIds": [
          "research-automation/fact/mlrc/gap-closed"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Spesifikk agent og syv konkurranseoppgaver."
        ],
        "approvedAt": "2026-08-13T09:38:37.210Z"
      },
      {
        "id": "research-automation/claim/mlrc-scope",
        "claimType": "method",
        "statement": "MLRC-Bench bruker syv dynamiske ML-konkurranseoppgaver med objektive resultatmål.",
        "asOf": "2025-04-13",
        "reviewState": "reviewed",
        "sourceIds": [
          "mlrc-bench-2025"
        ],
        "measurementIds": [
          "research-automation/mlrc/task-count"
        ],
        "factIds": [
          "research-automation/fact/mlrc/tasks"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Smal ML-domene og liten suite."
        ],
        "approvedAt": "2026-08-13T09:38:37.210Z"
      },
      {
        "id": "research-automation/claim/experimental-bottleneck",
        "claimType": "derivation",
        "statement": "De tre benchmarkene peker samlet mot implementasjon og eksperimentell validitet som en tydelig begrensning, men de kan ikke summeres til én automatiseringsandel.",
        "asOf": "2025-06-27",
        "reviewState": "reviewed",
        "sourceIds": [
          "rexbench-2025",
          "mlr-bench-2025",
          "mlrc-bench-2025"
        ],
        "measurementIds": [
          "research-automation/rexbench/best-with-hints",
          "research-automation/mlr/invalid-experiments",
          "research-automation/mlrc/gap-closed"
        ],
        "factIds": [
          "research-automation/fact/rexbench/best-with-hints",
          "research-automation/fact/mlr/invalid-results",
          "research-automation/fact/mlrc/gap-closed"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Kvalitativ triangulering, ikke en metaanalyse eller felles score."
        ],
        "approvedAt": "2026-08-13T09:38:37.210Z"
      },
      {
        "id": "research-automation/claim/no-aggregate",
        "claimType": "limitation",
        "statement": "Modulen publiserer ingen samlet FoU-automatiseringsprosent fordi benchmarkene ikke måler samme oppgave, agent eller suksesskriterium.",
        "asOf": "2026-08-12",
        "reviewState": "reviewed",
        "sourceIds": [
          "rexbench-2025",
          "mlr-bench-2025",
          "mlrc-bench-2025"
        ],
        "measurementIds": [
          "research-automation/workflow/coverage"
        ],
        "factIds": [
          "research-automation/fact/coverage/no-aggregate"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Et målegap, ikke bevis for null automatisering."
        ],
        "approvedAt": "2026-08-13T09:38:37.210Z"
      },
      {
        "id": "research-automation/claim/validation-reproduction-gap",
        "claimType": "limitation",
        "statement": "De nåværende benchmarkene etablerer ikke en full, uavhengig validerings- og reproduksjonssløyfe etter eksperiment og tolkning.",
        "asOf": "2026-08-22",
        "reviewState": "reviewed",
        "sourceIds": [
          "mlr-bench-2025",
          "rexbench-2025",
          "mlrc-bench-2025"
        ],
        "measurementIds": [
          "research-automation/workflow/validation-reproduction-gap"
        ],
        "factIds": [
          "research-automation/fact/coverage/validation-reproduction-gap"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Et dekningsgap i de valgte benchmarkene, ikke bevis for at slik validering aldri skjer."
        ],
        "approvedAt": "2026-08-22T17:16:21.000Z"
      },
      {
        "id": "research-automation/claim/hypothesis-stage-evidence",
        "claimType": "method",
        "statement": "Hypotesefasen har delvis benchmarkdekning: MLR-Bench inkluderer forslag i sin arbeidsflyt, mens RExBench avgrenser tolv forskningsutvidelser som konkrete oppgaver.",
        "asOf": "2025-06-27",
        "reviewState": "reviewed",
        "sourceIds": [
          "mlr-bench-2025",
          "rexbench-2025"
        ],
        "measurementIds": [
          "research-automation/mlr/workflow-scope",
          "research-automation/rexbench/task-count"
        ],
        "factIds": [
          "research-automation/fact/mlr/stages",
          "research-automation/fact/rexbench/tasks"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Kildene bruker ulike oppgaver og evaluatorer og etablerer ikke en felles suksessrate."
        ],
        "approvedAt": "2026-08-22T17:16:21.000Z"
      },
      {
        "id": "research-automation/claim/experiments-stage-evidence",
        "claimType": "method",
        "statement": "MLR-Bench inkluderer eksperimentering i arbeidsflyten, men rapporterer samtidig alvorlige invaliderte resultater for ett testet agentoppsett.",
        "asOf": "2025-05-26",
        "reviewState": "reviewed",
        "sourceIds": [
          "mlr-bench-2025"
        ],
        "measurementIds": [
          "research-automation/mlr/invalid-experiments",
          "research-automation/mlr/workflow-scope"
        ],
        "factIds": [
          "research-automation/fact/mlr/invalid-results",
          "research-automation/fact/mlr/stages"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Dekning av eksperimentfasen er ikke det samme som pålitelig eksperimentell kontroll."
        ],
        "approvedAt": "2026-08-22T17:16:21.000Z"
      },
      {
        "id": "research-automation/claim/evidence-state",
        "claimType": "method",
        "statement": "Research Automation har tre kildebelagte benchmarkfamilier, men null longitudinelle punkter innen samme benchmarkversjon. RExBench, MLR-Bench og MLRC-Bench har ulike oppgaver, evaluatorer og nevnere og etablerer derfor ingen felles trend.",
        "asOf": "2026-08-21",
        "reviewState": "reviewed",
        "sourceIds": [
          "rexbench-2025",
          "mlr-bench-2025",
          "mlrc-bench-2025",
          "research-automation-rexbench-paper-06e0d5161012"
        ],
        "measurementIds": [
          "research-automation/rexbench/best-with-hints",
          "research-automation/mlr/invalid-experiments",
          "research-automation/mlrc/gap-closed"
        ],
        "factIds": [],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Benchmarkfamiliene kan ikke summeres til én automatiseringsandel.",
          "Modenhet beskriver evidensdekning, ikke generell forskningsautonomi."
        ],
        "approvedAt": "2026-08-22T17:16:21.000Z"
      }
    ],
    "claimHistory": [
      {
        "id": "research-automation/claim/rexbench-capability",
        "claimType": "observation",
        "statement": "RExBench rapporterer at beste resultat, selv med menneskeskrevne hints, er under 40 % på realistiske forskningsutvidelser.",
        "asOf": "2025-06-27",
        "reviewState": "reviewed",
        "sourceIds": [
          "rexbench-2025"
        ],
        "measurementIds": [
          "research-automation/rexbench/best-with-hints"
        ],
        "factIds": [
          "research-automation/fact/rexbench/best-with-hints"
        ],
        "lineageSetIds": [],
        "methodologyPaths": [
          "data/tracker/research-automation/METHODOLOGY.md"
        ],
        "calculation": null,
        "limitations": [
          "Avhengig av oppsett og hint."
        ],
        "revisionId": "research-automation/claim/rexbench-capability@7b85e6ff0b18",
        "supersededAt": "2026-08-13T09:42:11.000Z",
        "supersededByProposalId": "proposal.research-automation.06e0d5161012.rexbench-v3-result-correction-fingerprint-v2"
      }
    ]
  },
  "drilldowns": {
    "schemaVersion": 1,
    "moduleId": "research-automation",
    "views": [
      {
        "schemaVersion": 1,
        "id": "benchmarks",
        "moduleId": "research-automation",
        "title": "Hva kan forskningsagentene faktisk gjøre?",
        "shortTitle": "AI-FoU",
        "description": "Tre komplementære benchmarklinser: implementere forskningsutvidelser, foreslå og teste nye metoder, og gjennomføre en åpen forskningsarbeidsflyt.",
        "decisionQuestion": "Hvor i forskningsløpet finnes demonstrert kapasitet, og hvor er resultatene fortsatt upålitelige?",
        "decisionUse": "Bruk kartet til å skille imponerende delresultater fra pålitelig, objektivt verifisert ende-til-ende-forskning.",
        "asOf": "2025-06-27",
        "sourceIds": [
          "rexbench-2025",
          "mlr-bench-2025",
          "mlrc-bench-2025",
          "research-automation-rexbench-paper-06e0d5161012"
        ],
        "headline": {
          "label": "Beste MLRC-Bench-gap lukket",
          "value": "9,3 %",
          "detail": "Objektivt målt, men på bare syv konkurranseoppgaver.",
          "classification": "observation",
          "claimId": "research-automation/claim/mlrc-capability"
        },
        "visualization": {
          "id": "research-capability-landscape",
          "type": "research-loop-coverage",
          "eyebrow": "RESEARCH LOOP COVERAGE / HVA ER FAKTISK TESTET?",
          "title": "Forskningsløkken har dekning, men ikke lukket validering",
          "description": "Prosesskartet viser hvor dagens benchmarkevidens treffer forskningsløkken. Benchmarkfeltene beholder sine egne nevnere og kan ikke leses som en rangering.",
          "measurementIds": [
            "research-automation/rexbench/best-with-hints",
            "research-automation/rexbench/task-count",
            "research-automation/mlrc/gap-closed",
            "research-automation/mlrc/task-count",
            "research-automation/mlr/invalid-experiments",
            "research-automation/mlr/workflow-scope",
            "research-automation/workflow/validation-reproduction-gap"
          ],
          "claimIds": [
            "research-automation/claim/rexbench-capability",
            "research-automation/claim/rexbench-scope",
            "research-automation/claim/mlrc-capability",
            "research-automation/claim/mlrc-scope",
            "research-automation/claim/mlr-reliability",
            "research-automation/claim/mlr-scope",
            "research-automation/claim/hypothesis-stage-evidence",
            "research-automation/claim/experiments-stage-evidence",
            "research-automation/claim/no-aggregate",
            "research-automation/claim/validation-reproduction-gap"
          ],
          "stages": [
            {
              "id": "problem-selection",
              "label": "Problemvalg",
              "status": "partial",
              "evidence": "MLR-Bench inkluderer idéstadiet, men innen en avgrenset benchmarkramme.",
              "measurementIds": [
                "research-automation/mlr/workflow-scope"
              ],
              "claimId": "research-automation/claim/mlr-scope"
            },
            {
              "id": "hypothesis",
              "label": "Hypotese / forslag",
              "status": "partial",
              "evidence": "MLR-Bench inkluderer forslag; RExBench gir avgrensede forskningsutvidelser.",
              "measurementIds": [
                "research-automation/mlr/workflow-scope",
                "research-automation/rexbench/task-count"
              ],
              "claimId": "research-automation/claim/hypothesis-stage-evidence"
            },
            {
              "id": "implementation",
              "label": "Implementasjon",
              "status": "partial",
              "evidence": "RExBench tester realistiske utvidelser, men beste resultat med hints er under 44 prosent.",
              "measurementIds": [
                "research-automation/rexbench/best-with-hints"
              ],
              "claimId": "research-automation/claim/rexbench-capability"
            },
            {
              "id": "experiments",
              "label": "Eksperimenter",
              "status": "risk",
              "evidence": "MLR-Bench dekker eksperimentering, men rapporterer alvorlige validitetsfeil i et testet agentoppsett.",
              "measurementIds": [
                "research-automation/mlr/invalid-experiments",
                "research-automation/mlr/workflow-scope"
              ],
              "claimId": "research-automation/claim/experiments-stage-evidence"
            },
            {
              "id": "interpretation",
              "label": "Tolkning / rapport",
              "status": "partial",
              "evidence": "MLR-Bench inkluderer paperskriving; dette etablerer ikke uavhengig vitenskapelig kontroll.",
              "measurementIds": [
                "research-automation/mlr/workflow-scope"
              ],
              "claimId": "research-automation/claim/mlr-scope"
            },
            {
              "id": "validation",
              "label": "Validering / reproduksjon",
              "status": "gap",
              "evidence": "Ingen opptatt benchmarkbane dokumenterer hele den uavhengige sluttkontrollen.",
              "measurementIds": [
                "research-automation/workflow/validation-reproduction-gap"
              ],
              "claimId": "research-automation/claim/validation-reproduction-gap"
            }
          ],
          "lanes": [
            {
              "id": "rexbench",
              "benchmark": "RExBench",
              "focus": "Forskningsutvidelser",
              "metricLabel": "Beste resultat med menneskeskrevne hints",
              "value": 44,
              "displayValue": "<44 %",
              "direction": "capability",
              "scope": "12 oppgaver",
              "measurementId": "research-automation/rexbench/best-with-hints",
              "claimId": "research-automation/claim/rexbench-capability"
            },
            {
              "id": "mlrc-bench",
              "benchmark": "MLRC-Bench",
              "focus": "Nye ML-metoder",
              "metricLabel": "Andel av gapet til toppmenneske som ble lukket",
              "value": 9.3,
              "displayValue": "9,3 %",
              "direction": "capability",
              "scope": "7 oppgaver",
              "measurementId": "research-automation/mlrc/gap-closed",
              "claimId": "research-automation/claim/mlrc-capability"
            },
            {
              "id": "mlr-bench",
              "benchmark": "MLR-Bench",
              "focus": "Eksperimentell kontroll",
              "metricLabel": "Eksempelvis ugyldige resultater i ett agentoppsett",
              "value": 80,
              "displayValue": "80 %",
              "direction": "risk",
              "scope": "201 oppgaver / 4 steg",
              "measurementId": "research-automation/mlr/invalid-experiments",
              "claimId": "research-automation/claim/mlr-reliability"
            }
          ],
          "trendBoundary": "Kildene er publisert på ulike datoer, men måler ulike oppgaver, evaluatorer og nevnere. Modulen har derfor ennå ingen forsvarlig longitudinal ytelsestrend; neste sammenlignbare benchmarkrelease må legges til i samme bane før en trend tegnes."
        },
        "sections": [
          {
            "id": "research-extension",
            "label": "IMPLEMENTASJON",
            "question": "Kan agenter utvide eksisterende forskning?",
            "conclusion": "RExBench v3 viser under 44 % selv med menneskeskrevne hints på 12 realistiske utvidelser.",
            "status": "constraint",
            "measurementIds": [
              "research-automation/rexbench/best-with-hints",
              "research-automation/rexbench/task-count"
            ],
            "claimIds": [
              "research-automation/claim/rexbench-capability",
              "research-automation/claim/rexbench-scope"
            ]
          },
          {
            "id": "novel-methods",
            "label": "NYE METODER",
            "question": "Kan agenter konkurrere mot toppmennesker på nye ML-problemer?",
            "conclusion": "Beste MLRC-Bench-agent lukket 9,3 % av gapet på syv objektivt evaluerte konkurranseoppgaver.",
            "status": "constraint",
            "measurementIds": [
              "research-automation/mlrc/gap-closed",
              "research-automation/mlrc/task-count"
            ],
            "claimIds": [
              "research-automation/claim/mlrc-capability",
              "research-automation/claim/mlrc-scope"
            ]
          },
          {
            "id": "end-to-end",
            "label": "ARBEIDSFLYT",
            "question": "Dekkes et helt forskningsløp?",
            "conclusion": "MLR-Bench dekker 201 oppgaver og fire steg, men rapporterer alvorlige problemer med eksperimentell validitet.",
            "status": "signal",
            "measurementIds": [
              "research-automation/mlr/workflow-scope",
              "research-automation/mlr/invalid-experiments"
            ],
            "claimIds": [
              "research-automation/claim/mlr-scope",
              "research-automation/claim/mlr-reliability"
            ]
          },
          {
            "id": "bottleneck",
            "label": "FELLES SIGNAL",
            "question": "Hva begrenser pålitelig autonom FoU?",
            "conclusion": "På tvers av ulike evalueringsformer er implementasjon og eksperimentell kontroll fortsatt tydelige svakheter.",
            "status": "constraint",
            "measurementIds": [
              "research-automation/rexbench/best-with-hints",
              "research-automation/mlrc/gap-closed",
              "research-automation/mlr/invalid-experiments"
            ],
            "claimIds": [
              "research-automation/claim/experimental-bottleneck"
            ]
          },
          {
            "id": "aggregate-gap",
            "label": "MÅLEGAP",
            "question": "Hvor stor andel av FoU er automatisert?",
            "conclusion": "Det finnes ingen forsvarlig felles nevner for en samlet automatiseringsprosent.",
            "status": "gap",
            "measurementIds": [
              "research-automation/workflow/coverage"
            ],
            "claimIds": [
              "research-automation/claim/no-aggregate"
            ]
          }
        ],
        "comparisons": [],
        "limitations": [
          "Resultatene er ikke tidsseriekompatible.",
          "Forskjellige evaluatorer gir forskjellige feilkilder.",
          "Ingen av benchmarkene observerer hemmelig eller kommersiell laboratoriepraksis."
        ]
      }
    ]
  },
  "sources": [
    {
      "id": "rexbench-2025",
      "title": "RExBench: Can coding agents autonomously implement AI research extensions?",
      "publisher": "Edwards et al.",
      "sourceUrl": "https://arxiv.org/abs/2506.22598",
      "documentationUrl": "https://arxiv.org/abs/2506.22598",
      "license": "Cited public paper; no paper text redistributed",
      "retrievedAt": "2026-08-12",
      "publishedAt": "2025-06-27",
      "dataAsOf": "2025-06-27",
      "sha256": null,
      "localPath": null,
      "role": "benchmark-result",
      "notes": "Twelve research-extension tasks and reported agent evaluation; do not generalize to all research work."
    },
    {
      "id": "mlr-bench-2025",
      "title": "MLR-Bench: Evaluating AI Agents on Open-Ended Machine Learning Research",
      "publisher": "Chen et al.",
      "sourceUrl": "https://arxiv.org/abs/2505.19955",
      "documentationUrl": "https://arxiv.org/abs/2505.19955",
      "license": "Cited public paper; no paper text redistributed",
      "retrievedAt": "2026-08-12",
      "dataAsOf": "2025-05-26",
      "sha256": null,
      "localPath": null,
      "role": "benchmark-result",
      "notes": "Open-ended ML research benchmark; the reported coding-agent failure example is not a cross-benchmark rate."
    },
    {
      "id": "mlrc-bench-2025",
      "title": "MLRC-Bench: Can Language Agents Solve Machine Learning Research Challenges?",
      "publisher": "Zhang et al.",
      "sourceUrl": "https://arxiv.org/abs/2504.09702",
      "documentationUrl": "https://arxiv.org/abs/2504.09702",
      "license": "Cited public paper; no paper text redistributed",
      "retrievedAt": "2026-08-12",
      "dataAsOf": "2025-04-13",
      "sha256": null,
      "localPath": null,
      "role": "benchmark-result",
      "notes": "Seven objectively scored ML research competition tasks; kept separate from RExBench and MLR-Bench."
    },
    {
      "id": "research-automation-rexbench-paper-06e0d5161012",
      "title": "RExBench paper",
      "publisher": "Edwards et al.",
      "sourceUrl": "https://arxiv.org/abs/2506.22598",
      "documentationUrl": "https://arxiv.org/abs/2506.22598",
      "license": "citation-only-review-required",
      "retrievedAt": "2026-08-13",
      "publishedAt": "2026-04-21",
      "dataAsOf": "2026-04-21",
      "sha256": "06e0d516101236b7e3a33c87da2d18d98949897da9339ace51f0d2cf37a122f3",
      "localPath": null,
      "role": "reviewed-research-source",
      "notes": "The current arXiv v3 abstract, revised 2026-04-21, reports about 33% autonomous success and says the best hinted result remains below 44%. This supersedes the legacy candidate whose hash represented the identical raw page bytes rather than the normalized visible-content fingerprint required by source admission.",
      "admission": {
        "candidateId": "research.research-automation.2026-08-13.rexbench-v3-result-correction-fingerprint-v2",
        "admittedAt": "2026-08-13T09:41:30.222Z",
        "contentFingerprint": "e4b79aa9be447bdf2707d492f0753bcb64597ec6b58e033c9853f8dcd9713e57",
        "rawSha256": "06e0d516101236b7e3a33c87da2d18d98949897da9339ace51f0d2cf37a122f3",
        "mediaType": "text/html",
        "finalUrl": "https://arxiv.org/abs/2506.22598",
        "archiveException": {
          "rationale": "ArXiv abstract-page evidence is citation-only in this tracker: retain its URL, normalized fingerprint, and raw-byte SHA-256 without redistributing the page bytes; owner accepted this bounded exception on 2026-08-13.",
          "acceptedBy": "owner/kristian",
          "acceptedAt": "2026-08-13T09:41:22.184Z"
        }
      }
    }
  ]
}