{
  "schema_version": "1.0",
  "snapshot": "e84e3b8153c8042bd823cf2760cfc355fb003218",
  "registration_state": "draft-slate",
  "license_scope": "Each record’s CC BY 4.0 licence covers that record’s text and metadata. Each cited source keeps its own terms.",
  "records": [
    {
      "record": {
        "schema_version": "1.0",
        "id": "clm-2026-0001",
        "type": "claim",
        "title": "Bounded self-improvement",
        "status": "candidate",
        "verdict": "supported_within_scope",
        "evidence_mode": "source_report",
        "verification_status": "source_support_checked",
        "inference_strength": "observed",
        "source_kinds": [
          "research paper"
        ],
        "method_kinds": [
          "reported experiment"
        ],
        "verified_on": "2026-09-24",
        "verifier": "Hapax Research Labs source review",
        "statement": "The Darwin Gödel Machine authors report benchmark gains from iterative code modification and evaluation, with sandboxing and human oversight.",
        "evidence_scope": "Coding benchmarks in the reported Darwin Gödel Machine experiment; not deployment-wide or economy-wide.",
        "sources": [
          "src-dgm-v3"
        ],
        "limitations": "This supports bounded self-improvement; it does not establish an autonomous, general research loop or independent replication by this lab.",
        "license": "CC-BY-4.0"
      },
      "body": "The next question is whether the reported gains transfer to a different task distribution under a declared evaluation protocol. This source check alone does not answer it.",
      "sha256": "1c79c943d2911cbde6b25e24e4e568a45ba033a48830db5e88888f3b7b08bdde"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "clm-2026-0002",
        "type": "claim",
        "title": "Feedback dynamics remain model-dependent",
        "status": "candidate",
        "verdict": "open",
        "evidence_mode": "model_estimate",
        "verification_status": "source_support_checked",
        "inference_strength": "estimated",
        "source_kinds": [
          "research paper"
        ],
        "method_kinds": [
          "economic modelling",
          "derivation"
        ],
        "verified_on": "2026-09-24",
        "verifier": "Hapax Research Labs source review",
        "statement": "Cunningham and colleagues’ calibration suggests current feedback is below self-sustaining acceleration while strengthening. Separately, Burtsev derives conditions under which amplification can precede visible acceleration.",
        "evidence_scope": "Specified economic models and their assumptions; not an observation of a deployed threshold crossing.",
        "sources": [
          "src-feedback-v1",
          "src-reproduction-v1"
        ],
        "limitations": "These are model-dependent conclusions, not an exhaustive finding that no contrary evidence exists, or an observation that a threshold has been crossed.",
        "license": "CC-BY-4.0"
      },
      "body": "Checking the published description does not validate its assumptions. A change in the measured quantities on which the models depend would require a new assessment, with its own scope and source check.",
      "sha256": "67836a74bb8e2578ec2f581823f521918ad07223e467639b03560c1aec7e3766"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "cor-2026-0001",
        "type": "correction",
        "title": "Feedback dynamics remain model-dependent (corrected wording)",
        "status": "candidate",
        "prior": "clm-2026-0002",
        "original_wording": "Cunningham and colleagues’ calibration suggests current feedback is below self-sustaining acceleration while strengthening. Separately, Burtsev derives conditions under which amplification can precede visible acceleration.",
        "reason": "The source’s own wording marks this conclusion as a back-of-the-envelope calculation (arXiv 2609.15802v1, abstract), and the original statement dropped that marker.",
        "corrected_on": "2026-09-25",
        "verdict": "open",
        "evidence_mode": "model_estimate",
        "verification_status": "source_support_checked",
        "inference_strength": "estimated",
        "source_kinds": [
          "research paper"
        ],
        "method_kinds": [
          "economic modelling",
          "derivation"
        ],
        "verified_on": "2026-09-25",
        "verifier": "Hapax Research Lab source review",
        "statement": "Cunningham and colleagues’ back-of-the-envelope calibration suggests current feedback is below self-sustaining acceleration while strengthening. Separately, Burtsev derives conditions under which amplification can precede visible acceleration.",
        "evidence_scope": "Specified economic models and their assumptions; not an observation of a deployed threshold crossing.",
        "sources": [
          "src-feedback-v1",
          "src-reproduction-v1"
        ],
        "limitations": "These are model-dependent conclusions, not an exhaustive finding that no contrary evidence exists, or an observation that a threshold has been crossed.",
        "license": "CC-BY-4.0"
      },
      "body": "Checking the published description does not validate its assumptions. A change in the measured quantities on which the models depend would require a new assessment, with its own scope and source check.",
      "sha256": "55f791366934d046c0a69de80c45e11ab38756c13af7a8df59ed5ed3e4b22598"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "pred-2026-0001",
        "type": "prediction",
        "slate_id": "P1",
        "title": "EU delegated act",
        "statement": "The European Commission will not adopt, on or before 2 August 2027, a delegated act under Article 51(3) of Regulation (EU) 2024/1689 that either changes the numerical 10^25-FLOP threshold in Article 51(2) or supplements Article 51 with a benchmark or indicator for determining high-impact capabilities. Resolve by Commission adoption date, allowing 30 days for documentary publication; drafts, consultations, guidance, designations, implementing acts, and ordinary legislation do not count.",
        "status": "draft",
        "verdict": "open",
        "evidence_mode": "source_report",
        "evidence_scope": "HRL-inferred prospective forecast over public institutional records; no safety-outcome inference.",
        "verification_status": "not_verified",
        "inference_strength": "forecast",
        "source_kinds": [
          "public institutional records"
        ],
        "method_kinds": [
          "single-elicitor forecast"
        ],
        "sources": [
          "src-eu-article51"
        ],
        "verified_on": null,
        "verifier": null,
        "limitations": "Draft candidate, not public preregistration. Full resolution cards, source snapshots, attribution, baselines, independent checker, and publication admission are outstanding.",
        "probability": 0.4,
        "registered_at": null,
        "deadline": "2027-08-02T23:59:59Z",
        "resolution_criteria": "The European Commission will not adopt, on or before 2 August 2027, a delegated act under Article 51(3) of Regulation (EU) 2024/1689 that either changes the numerical 10^25-FLOP threshold in Article 51(2) or supplements Article 51 with a benchmark or indicator for determining high-impact capabilities. Resolve by Commission adoption date, allowing 30 days for documentary publication; drafts, consultations, guidance, designations, implementing acts, and ordinary legislation do not count.",
        "registration_receipt": null,
        "checker": null,
        "baseline_probability": null,
        "outcome": null,
        "resolved_at": null,
        "category": "assessor",
        "scoring_rule": "Brier; cluster-balanced and ordinary mean reported separately",
        "license": "CC-BY-4.0"
      },
      "body": "The proposition and candidate probability are reproduced from the frozen six-entry draft slate. Inclusion here does not register the prediction. All dates and resolution details require the full registration card before prospective scoring begins.",
      "sha256": "47daa5d48864dd21c675c6a22db215733b0fc0efa22ef645eecb33f9d82ccef3"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "pred-2026-0002",
        "type": "prediction",
        "slate_id": "P2",
        "title": "Public lab threshold attainment",
        "statement": "By 2027-09-30, Anthropic will expressly conclude that the “Automated R&D in key domains” threshold in RSP v3.4 has been met, or Google DeepMind will expressly conclude that a named model reached ML R&D Automation Level 1 or Acceleration Level 1 under FSF v3.1. Only explicit public attainment counts; alert thresholds, precautionary safeguards, or later framework renaming do not.",
        "status": "draft",
        "verdict": "open",
        "evidence_mode": "source_report",
        "evidence_scope": "HRL-inferred prospective forecast over public institutional records; no safety-outcome inference.",
        "verification_status": "not_verified",
        "inference_strength": "forecast",
        "source_kinds": [
          "public institutional records"
        ],
        "method_kinds": [
          "single-elicitor forecast"
        ],
        "sources": [
          "src-anthropic-rsp",
          "src-deepmind-fsf"
        ],
        "verified_on": null,
        "verifier": null,
        "limitations": "Draft candidate, not public preregistration. Full resolution cards, source snapshots, attribution, baselines, independent checker, and publication admission are outstanding.",
        "probability": 0.22,
        "registered_at": null,
        "deadline": "2027-09-30T23:59:59Z",
        "resolution_criteria": "By 2027-09-30, Anthropic will expressly conclude that the “Automated R&D in key domains” threshold in RSP v3.4 has been met, or Google DeepMind will expressly conclude that a named model reached ML R&D Automation Level 1 or Acceleration Level 1 under FSF v3.1. Only explicit public attainment counts; alert thresholds, precautionary safeguards, or later framework renaming do not.",
        "registration_receipt": null,
        "checker": null,
        "baseline_probability": null,
        "outcome": null,
        "resolved_at": null,
        "category": "assessor",
        "scoring_rule": "Brier; cluster-balanced and ordinary mean reported separately",
        "license": "CC-BY-4.0"
      },
      "body": "The proposition and candidate probability are reproduced from the frozen six-entry draft slate. Inclusion here does not register the prediction. All dates and resolution details require the full registration card before prospective scoring begins.",
      "sha256": "e901130b22884e154bc3ed6b31ee810542a5353f594ab5bd1b54245635b9a665"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "pred-2026-0003",
        "type": "prediction",
        "slate_id": "P3",
        "title": "Joint evaluation report",
        "statement": "By 2027-09-30, NAAIMES will publish or formally co-issue a joint evaluation report identifying at least three member institutes, naming at least three model versions from at least two developers, reporting a common numeric model-level metric for every named model, and disclosing the task set, scoring rule, model/version, and principal inference or elicitation settings.",
        "status": "draft",
        "verdict": "open",
        "evidence_mode": "source_report",
        "evidence_scope": "HRL-inferred prospective forecast over public institutional records; no safety-outcome inference.",
        "verification_status": "not_verified",
        "inference_strength": "forecast",
        "source_kinds": [
          "public institutional records"
        ],
        "method_kinds": [
          "single-elicitor forecast"
        ],
        "sources": [
          "src-naaimes"
        ],
        "verified_on": null,
        "verifier": null,
        "limitations": "Draft candidate, not public preregistration. Full resolution cards, source snapshots, attribution, baselines, independent checker, and publication admission are outstanding.",
        "probability": 0.42,
        "registered_at": null,
        "deadline": "2027-09-30T23:59:59Z",
        "resolution_criteria": "By 2027-09-30, NAAIMES will publish or formally co-issue a joint evaluation report identifying at least three member institutes, naming at least three model versions from at least two developers, reporting a common numeric model-level metric for every named model, and disclosing the task set, scoring rule, model/version, and principal inference or elicitation settings.",
        "registration_receipt": null,
        "checker": null,
        "baseline_probability": null,
        "outcome": null,
        "resolved_at": null,
        "category": "assessor",
        "scoring_rule": "Brier; cluster-balanced and ordinary mean reported separately",
        "license": "CC-BY-4.0"
      },
      "body": "The proposition and candidate probability are reproduced from the frozen six-entry draft slate. Inclusion here does not register the prediction. All dates and resolution details require the full registration card before prospective scoring begins.",
      "sha256": "37fe328fa31d1664b8c9622d89cda62d482dd087f09921fa151ec9654eefe9d4"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "pred-2026-0004",
        "type": "prediction",
        "slate_id": "P4",
        "title": "California statutory report",
        "statement": "By 2027-01-31, Cal OES will file or publicly post its first report required by California Business and Professions Code §22757.13(g). A zero-incident report counts; a portal, press release, or legislative description without the statutory report does not.",
        "status": "draft",
        "verdict": "open",
        "evidence_mode": "source_report",
        "evidence_scope": "HRL-inferred prospective forecast over public institutional records; no safety-outcome inference.",
        "verification_status": "not_verified",
        "inference_strength": "forecast",
        "source_kinds": [
          "public institutional records"
        ],
        "method_kinds": [
          "single-elicitor forecast"
        ],
        "sources": [
          "src-cal-oes"
        ],
        "verified_on": null,
        "verifier": null,
        "limitations": "Draft candidate, not public preregistration. Full resolution cards, source snapshots, attribution, baselines, independent checker, and publication admission are outstanding.",
        "probability": 0.92,
        "registered_at": null,
        "deadline": "2027-02-01T07:59:59Z",
        "resolution_criteria": "By 2027-01-31, Cal OES will file or publicly post its first report required by California Business and Professions Code §22757.13(g). A zero-incident report counts; a portal, press release, or legislative description without the statutory report does not.",
        "registration_receipt": null,
        "checker": null,
        "baseline_probability": null,
        "outcome": null,
        "resolved_at": null,
        "category": "assessor",
        "scoring_rule": "Brier; cluster-balanced and ordinary mean reported separately",
        "license": "CC-BY-4.0"
      },
      "body": "The proposition and candidate probability are reproduced from the frozen six-entry draft slate. Inclusion here does not register the prediction. All dates and resolution details require the full registration card before prospective scoring begins.",
      "sha256": "2a0eb0e3b299dbc09c1e25f3b877d9a155778826db594484fb7b81d89306c200"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "pred-2026-0005",
        "type": "prediction",
        "slate_id": "P5",
        "title": "A 24-hour task horizon",
        "statement": "By 2027-09-30, METR will publish a task-completion-time-horizon estimate for a named publicly available model with a P50 time horizon of at least 24 hours under TH1.1 or an explicitly bridged successor methodology. Internal or unreleased models and non-estimate lower bounds do not count.",
        "status": "draft",
        "verdict": "open",
        "evidence_mode": "source_report",
        "evidence_scope": "HRL-inferred prospective forecast over public institutional records; no safety-outcome inference.",
        "verification_status": "not_verified",
        "inference_strength": "forecast",
        "source_kinds": [
          "public institutional records"
        ],
        "method_kinds": [
          "single-elicitor forecast"
        ],
        "sources": [
          "src-metr"
        ],
        "verified_on": null,
        "verifier": null,
        "limitations": "Draft candidate, not public preregistration. Full resolution cards, source snapshots, attribution, baselines, independent checker, and publication admission are outstanding.",
        "probability": 0.68,
        "registered_at": null,
        "deadline": "2027-09-30T23:59:59Z",
        "resolution_criteria": "By 2027-09-30, METR will publish a task-completion-time-horizon estimate for a named publicly available model with a P50 time horizon of at least 24 hours under TH1.1 or an explicitly bridged successor methodology. Internal or unreleased models and non-estimate lower bounds do not count.",
        "registration_receipt": null,
        "checker": null,
        "baseline_probability": null,
        "outcome": null,
        "resolved_at": null,
        "category": "capability",
        "scoring_rule": "Brier; cluster-balanced and ordinary mean reported separately",
        "license": "CC-BY-4.0"
      },
      "body": "The proposition and candidate probability are reproduced from the frozen six-entry draft slate. Inclusion here does not register the prediction. All dates and resolution details require the full registration card before prospective scoring begins.",
      "sha256": "67c724470ced9ede1e2508090281b1ad0a507efea91c369ee2b4708febcf2cac"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "pred-2026-0006",
        "type": "prediction",
        "slate_id": "P6",
        "title": "Final evaluation practices",
        "statement": "By 2027-09-30, NIST or CAISI will publish a final, non-draft edition of NIST AI 800-2, Practices for Automated Benchmark Evaluations of Language Models, or an officially renumbered successor with substantially the same scope. Revised drafts and requests for comment do not count.",
        "status": "draft",
        "verdict": "open",
        "evidence_mode": "source_report",
        "evidence_scope": "HRL-inferred prospective forecast over public institutional records; no safety-outcome inference.",
        "verification_status": "not_verified",
        "inference_strength": "forecast",
        "source_kinds": [
          "public institutional records"
        ],
        "method_kinds": [
          "single-elicitor forecast"
        ],
        "sources": [
          "src-nist-800-2"
        ],
        "verified_on": null,
        "verifier": null,
        "limitations": "Draft candidate, not public preregistration. Full resolution cards, source snapshots, attribution, baselines, independent checker, and publication admission are outstanding.",
        "probability": 0.76,
        "registered_at": null,
        "deadline": "2027-09-30T23:59:59Z",
        "resolution_criteria": "By 2027-09-30, NIST or CAISI will publish a final, non-draft edition of NIST AI 800-2, Practices for Automated Benchmark Evaluations of Language Models, or an officially renumbered successor with substantially the same scope. Revised drafts and requests for comment do not count.",
        "registration_receipt": null,
        "checker": null,
        "baseline_probability": null,
        "outcome": null,
        "resolved_at": null,
        "category": "assessor",
        "scoring_rule": "Brier; cluster-balanced and ordinary mean reported separately",
        "license": "CC-BY-4.0"
      },
      "body": "The proposition and candidate probability are reproduced from the frozen six-entry draft slate. Inclusion here does not register the prediction. All dates and resolution details require the full registration card before prospective scoring begins.",
      "sha256": "7ef16dc40744f204bf20c07baeeef48fb53f1dbc0b5a13b997486ad7ee3633c2"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-anthropic-rsp",
        "type": "source",
        "title": "Anthropic responsible scaling policy",
        "url": "https://www.anthropic.com/responsible-scaling-policy",
        "locator": "RSP v3.4 threshold text must be archived and bound before registration.",
        "check_scope": "Candidate resolution source; current-version landing page is not an immutable policy snapshot.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "6ac3216f78c10e87c3124baef5763a93e799867c0cedf4e2864a69046249130a"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-cal-oes",
        "type": "source",
        "title": "California Office of Emergency Services",
        "url": "https://www.caloes.ca.gov/",
        "locator": "First report required by Business and Professions Code section 22757.13(g).",
        "check_scope": "Candidate source family; exact report archive and deadline timezone require registration-card verification.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "706a195ea9af817a35200e344bddb5de0e4b84282fddaab9f2fa450672545501"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-deepmind-fsf",
        "type": "source",
        "title": "Google DeepMind frontier safety framework",
        "url": "https://deepmind.google/discover/blog/introducing-the-frontier-safety-framework/",
        "locator": "FSF v3.1 threshold text must be archived and bound before registration.",
        "check_scope": "Candidate source; versioned policy and explicit public attainment statement remain required.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "c93333a26575a7c9d595daec449e211ad2cd412370aaea22e430d5c6a1fb48dc"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-dgm-v3",
        "type": "source",
        "title": "Darwin Gödel Machine — version 3",
        "url": "https://arxiv.org/abs/2505.22954v3",
        "locator": "Abstract: benchmark gains, iterative modification, sandboxing and oversight.",
        "check_scope": "Abstract support checked in the September 24 source review; no independent replication.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "9e12e4d5a2dea6da54652846c52d87c885a7c195de4202460a6d4cce56d3c3f6"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-eu-article51",
        "type": "source",
        "title": "EU AI Act — Article 51",
        "url": "https://eur-lex.europa.eu/eli/reg/2024/1689/oj/eng",
        "locator": "Article 51(2) and 51(3).",
        "check_scope": "Candidate resolution source; the future event has not been checked.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "29ecfb515a02d02c1bbcf11f31e99ac0fab0127932e376ffe7c707a20d876d6f"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-feedback-v1",
        "type": "source",
        "title": "Modelled feedback in AI research",
        "url": "https://arxiv.org/abs/2609.15802v1",
        "locator": "Version 1 abstract; calibration of feedback dynamics.",
        "check_scope": "Abstract support checked in the September 24 source review; assumptions and empirical validity are not independently established.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "b7114c89e733e31ce5694d6a5c564492140c26a020502e5ed8e6f857a9cfdd99"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-metr",
        "type": "source",
        "title": "METR time horizon measurements",
        "url": "https://metr.org/time-horizons/",
        "locator": "TH1.1 or an explicitly bridged successor; named public model P50 estimate.",
        "check_scope": "Candidate resolution source; draft benchmark qualification has not been independently checked for registration.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "b2849ebcb9634eac80c27e5cf874306c3bb4199de690337765cf244e01937e4a"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-naaimes",
        "type": "source",
        "title": "AI measurement and evaluation network",
        "url": "https://www.nist.gov/caisi",
        "locator": "Network publication and co-issuer evidence required by the proposition.",
        "check_scope": "Candidate source family; the exact authoritative network archive remains to be bound before registration.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "18aefbdef3f52cfd6599238252971c77dfac584878af951776cb312cc0d88e56"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-nist-800-2",
        "type": "source",
        "title": "NIST AI 800-2",
        "url": "https://doi.org/10.6028/NIST.AI.800-2.ipd",
        "locator": "Draft lineage only; resolution requires a final non-draft publication.",
        "check_scope": "Candidate source; a draft DOI is not evidence that the forecast has resolved.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "f3882d5b9af2d8d7d993a506ea01c3081061ef9034a04d1ef1c85ed4af444869"
    },
    {
      "record": {
        "schema_version": "1.0",
        "id": "src-reproduction-v1",
        "type": "source",
        "title": "Recursive reproduction and acceleration",
        "url": "https://arxiv.org/abs/2609.00137v1",
        "locator": "Version 1; theoretical conditions for amplification.",
        "check_scope": "Source cited by the September 24 narrative review; not an observation of a deployed threshold crossing.",
        "license": "CC-BY-4.0"
      },
      "body": "Inspect the source and the stated location. A resolving link does not establish that it supports every claim made about it.",
      "sha256": "323be0c128bf175e2caea9a3ffd3bfe99790d0ac6e69761921ca642b60af8459"
    }
  ]
}
