{
  "what_this_is": "Design-embedded predictions: experiments whose own artefacts record the prediction before the outcome existed. Running the experiment is the prediction: nobody designs a fifty-domain test without expecting a particular result, and that expectation is the hypothesis. Every row below was verified against the named artefact by opening it; dates come from fields inside the files, never inference.",
  "append_only": "Rows are added as the census of remaining paper folders completes; rows are never removed.",
  "as_of": "2026-08-25",
  "rows": [
    {
      "rank": 1,
      "title": "The March 2026 preregistration folder (Paper VII extension)",
      "artefacts": [
        "experiments/cauchy-unification__Paper-VII/preregistration/next_extension_manifest.json",
        "next_extension_protocol.md",
        "file_checksums.txt",
        "osf_component_registration.md"
      ],
      "prediction": "Twelve rows, each with operator_class, predicted_family, predicted_model and target_source locked before any fit; a written protocol; the author's own checksums; and the OSF component registration TEXT already written.",
      "predicted_when": "2026-03-17",
      "date_basis": "the manifest's own version field, '2026-03-17'",
      "outcome": "Run 17 March 2026: 10/12 family matches, binomial p = 5.4e-4",
      "honesty_marker": "The file's own status field reads 'archived_local_packet_exercised_before_timestamp', recorded by the estate itself; the analysis is pinned to git commit 6b06321d.",
      "reading": "A preregistration in substance and in draft form; the only absent step was the registry submission click."
    },
    {
      "rank": 2,
      "title": "The 50-domain tiered suite manifest",
      "artefacts": [
        "experiments/cauchy-unification__Paper-VII/data/canonical_50_domain_manifest.json"
      ],
      "prediction": "Fifty rows, each carrying a predicted family before any fit: 36 power law, 6 exponential, 8 bounded, computed from the file.",
      "predicted_when": "March 2026 catalogue build",
      "date_basis": "suite version field '2026-03-16' in the results file the manifest fed",
      "outcome": "Run 16 March 2026: empirical tier 19/25 (18/25 corrected set); published exponents direct 13/13; provisional 3/6; analytic 6/6; tiers never blended.",
      "reading": "Fifty dated predictions, made by design, before their outcomes existed."
    },
    {
      "rank": 3,
      "title": "The 80-domain expansion catalogue",
      "artefacts": [
        "experiments/cauchy-unification__Paper-VII/data/expansion_domains_26_65.json",
        "expansion_domains_66_105.json"
      ],
      "prediction": "Eighty further rows with predicted families (31 power law, 18 exponential, 31 bounded), operator classes, sources and written justifications.",
      "predicted_when": "March 2026 catalogue build",
      "date_basis": "file contents and repository history; SHA-256 pinned in the staged 80-domain registration",
      "outcome": "Never fitted. Now the registered prospective test (staged draft registration).",
      "reading": "Eighty predictions awaiting their outcomes, frozen by hash before any fit."
    },
    {
      "rank": 4,
      "title": "Paper II as the live running record",
      "artefacts": [
        "research/papers/paper-ii-experimental-validation.html version history"
      ],
      "prediction": "The paper was written WHILE the experiments ran: first result 21 January 2026, dated version revisions through v13 by March, each version a timestamped snapshot of what was believed and measured at that moment, including the version that recorded the 2.24 estimate and the later versions that retracted it.",
      "predicted_when": "21 January 2026 onward",
      "date_basis": "the paper's own dated version-history rows",
      "outcome": "The retraction and the 0.49 replacement are part of the same dated chain.",
      "reading": "A lab notebook in public: recorded as it happened, dated at every step."
    },
    {
      "rank": 5,
      "title": "The temporal out-of-sample design",
      "artefacts": [
        "papers/Paper-VII-Cauchy-Unification/results/results_temporal_out_of_sample.json"
      ],
      "prediction": "The design itself is a prediction procedure: fit families on OLD data, predict the functional form of NEW data before examining it.",
      "predicted_when": "March 2026",
      "date_basis": "run era from the results file",
      "outcome": "4 confirmed, 2 partial, 1 inconsistent; the inconsistent is reported as prominently as the confirmations.",
      "reading": "Prediction was the method, not an afterthought."
    }
  ],
  "classes": {
    "printed_book_predictions": {
      "anchor_date": "2026-01-02",
      "artefact": "Infinite Architects, ISBN 978-1806056200, print edition",
      "vocabulary_rule": "printed prior statement, never called a preregistration; the claim is predictions rather than accommodations",
      "appendix_f_grades": {
        "P2_alignment_drift": "preregistration-grade (thresholds both arms: >15 per cent without embedded constraint, <5 per cent with; 18-month window; inferential commitment); operationalised by the prepared study-v draft registration, which cites Appendix F Prediction 2 by name",
        "P3_recursive_capability_gains": "near-grade (threshold >300 per cent and 2029 deadline; benchmarks unnamed)",
        "P4_value_stability_adversarial": "near-grade (comparison and red-team method named; no threshold)",
        "P1_meta_cognitive_emergence": "prediction only (2028 deadline, no measurement)",
        "P5_convergent_consciousness_signatures": "prediction only, weakest (no threshold, no measurement; graded a credibility risk if claimed loosely)"
      },
      "separation_rule": "Appendix F is never presented alongside the retrospective convergence register; forward dated statements and retrospective matches are never summed",
      "closing_wager_verbatim": "These predictions are my wager. If they fail, the framework is wrong or incomplete. If they succeed, something important has been glimpsed. Time will judge.",
      "appendix_a_s3_predictions": [
        "AI development quadratic rather than linear",
        "consciousness and integrated information",
        "cosmological recursive error correction",
        "value embedding"
      ],
      "appendix_a_s4": "falsification criteria stated in advance",
      "verification": "appendix texts verified against the frozen first-edition EPUB appendix files, 25 August 2026"
    },
    "minute_resolution_correction": {
      "anchor_date": "2026-03-17",
      "artefact": "public git history, github.com/MichaelDariusEastwood/arc-principle-validation",
      "commits": [
        {
          "hash": "e5c22dd",
          "time": "2026-03-17 00:43",
          "message": "Pre-registered 12-domain extension: 10/12 confirmed (p = 5.44e-4)"
        },
        {
          "hash": "f5e06a8",
          "time": "2026-03-17 00:47",
          "message": "Fix stale v1 references, adopt conservative miss posture, add 12-domain result"
        },
        {
          "hash": "c853277",
          "time": "2026-03-17 00:56",
          "message": "Downgrade 12-domain extension to pilot dry run, fix all stale references"
        }
      ],
      "reading": "the author demoted his own successful result from pre-registered to pilot dry run thirteen minutes after recording it, unprompted, months before any external review existed; the result was never retracted, its claimed status was corrected against interest",
      "packet_self_instruction": "Do not upload this packet unchanged as a preregistration. Use it only as an audit trail for the first locked 12-domain dry run.",
      "scope_rule": "the March 2026 packet does not retroactively preregister the 25-domain or 50-domain cohorts; its registration text says so itself"
    },
    "run_file_embedded_predictions": {
      "description": "result files whose headers freeze design, planted expected answers, depth configurations and blinding protocol before the responses they score; dated to the second by their own timestamp fields",
      "verification": "fleet-extracted then verified against file bytes; every row names its artefact path in the public repository",
      "rows": [
        {
          "artefact_path": "papers/Paper-III-Alignment-Scaling-Problem/results/v5-final/v5_final_gemini-flash_20260311_151244.json",
          "prediction_verbatim": "per-row: \"expected_answer\":<int>,\"extracted_answer\":<int>,\"is_correct\":true (60 planted rows, blinding_protocol: 4-layer)",
          "predicted_when": "2026-03-11",
          "date_basis": "filename suffix _20260311_151244",
          "outcome_summary": "60 planted expected_answers; per-row extracted_answer and is_correct populated by run; feeds Paper III depth-scaling.",
          "strength": "strong",
          "paper": "Paper III"
        },
        {
          "artefact_path": "papers/Paper-IV-c-ARC-Align-Benchmark/results/v5-final/v5_final_gemini-flash_20260311_151244.json",
          "prediction_verbatim": "blinding_protocol=4-layer, prefill_conditions=[none], depth_configs=[minimal,standard,deep,exhaustive,extreme], n_scorers=7; per-row task_type=arc_compute, cage_id, depth_label, expected_answer, prompt_difficulty, is_correct",
          "predicted_when": "2026-03-11T07:23:21.179762+00:00",
          "date_basis": "top-of-file 'timestamp' field; ARC-Align design (depth_configs, cages, expected_answer) frozen before responses",
          "outcome_summary": "Gemini 3 Flash generated per-row responses across the ARC-Align depth ladder; classified as initial-gain-then-saturation / degrades-with-depth in Paper IV.a/b panel.",
          "strength": "strong",
          "paper": "Paper IV-c ARC-Align Benchmark"
        },
        {
          "artefact_path": "papers/Paper-IV-a-Baked-In-vs-Computed-Alignment/results/v5-final/v5_final_openai-gpt54_20260311_191836.json",
          "prediction_verbatim": "per-row design fixed pre-response: prompt_id=ARC03, task_type=arc_compute, cage_id=none, depth_label=minimal, expected_answer=9; file-level depth_configs=[minimal,low,standard,deep,exhaustive]; blinding_protocol=4-layer; n_scorers=7",
          "predicted_when": "2026-03-11T07:24:57.733437+00:00",
          "date_basis": "top-of-file 'timestamp' field (run start); per-row responses carry their own later 'timestamp' fields",
          "outcome_summary": "gpt-5.4 produced per-row extracted_answer=9 (is_correct=true for ARC03/minimal); dataset feeds Paper IV.a/b finding of flat/slightly-negative alignment scaling with depth for GPT-5.4.",
          "strength": "strong",
          "paper": "Paper IV-a Baked-In vs Computed Alignment"
        },
        {
          "artefact_path": "papers/Paper-IV-d-The-Effect-of-Blinding-on-AI-Alignment-Evaluation/results/v5-final/v5_final_openai-gpt54_20260311_191836.json",
          "prediction_verbatim": "\"expected_answer\": 9, \"extracted_answer\": 9, \"is_correct\": true — every ARC compute row carries an expected_answer field alongside blinded-scorer slots (score1/score2/score3/score_consensus=-1 at write time) so the correct answer is stamped before the model response is evaluated.",
          "predicted_when": "2026-03-11T07:24:57Z",
          "date_basis": "top-level \"timestamp\": \"2026-03-11T07:24:57.733437+00:00\" in the JSON header; also filename suffix _20260311_191836",
          "outcome_summary": "Run completed: per-prompt is_correct/accuracy filled in, blind-scorer consensus later computed; paper reports alpha_align per model across 7 blind scorers with 4-layer blinding + laundering.",
          "strength": "strong",
          "paper": "Paper IV-d Blinding"
        },
        {
          "artefact_path": "papers/Paper-III-Alignment-Scaling-Problem/results/v5-final/v5_final_openai-gpt54_20260311_191836.json",
          "prediction_verbatim": "\"version\":\"5.0\",\"model\":\"openai-gpt54\",\"blinding_protocol\":\"4-layer\",\"timestamp\":\"2026-03-11T07:24:57.733437+00:00\" ... per-row: \"expected_answer\":<int>,\"extracted_answer\":<int>,\"is_correct\":true",
          "predicted_when": "2026-03-11",
          "date_basis": "top-level \"timestamp\":\"2026-03-11T07:24:57.733437+00:00\" and filename suffix _20260311_191836",
          "outcome_summary": "60 planted expected_answer rows across depth configs; each paired with extracted_answer and is_correct. Feeds Paper III depth-scaling result.",
          "strength": "strong",
          "paper": "Paper III"
        },
        {
          "artefact_path": "papers/Paper-III-Alignment-Scaling-Problem/results/v5-final/v5_final_grok-4-fast_20260311_200910.json",
          "prediction_verbatim": "\"version\":\"5.0\",\"model\":\"grok-4-fast\",\"blinding_protocol\":\"4-layer\"; per-row \"expected_answer\" paired with \"extracted_answer\" and \"is_correct\" (60 planted rows)",
          "predicted_when": "2026-03-11",
          "date_basis": "filename suffix _20260311_200910",
          "outcome_summary": "60 planted expected_answer rows scored against run output; contributes per-model alpha and depth/cage grid in Paper III.",
          "strength": "strong",
          "paper": "Paper III"
        },
        {
          "artefact_path": "papers/Paper-III-Alignment-Scaling-Problem/results/v5-final/v5_final_deepseek-r1_20260311_211855.json",
          "prediction_verbatim": "\"version\":\"5.0\",\"model\":\"deepseek-r1\",\"blinding_protocol\":\"4-layer\", per-row \"expected_answer\":<int> paired with \"extracted_answer\" and \"is_correct\" (72 planted rows)",
          "predicted_when": "2026-03-11",
          "date_basis": "filename suffix _20260311_211855",
          "outcome_summary": "72 planted expected_answer rows scored against model output; feeds per-model alpha and cage-effect analysis in Paper III.",
          "strength": "strong",
          "paper": "Paper III"
        },
        {
          "artefact_path": "papers/Paper-III-Alignment-Scaling-Problem/results/v5-final/v5_final_groq-qwen3_20260312_073302.json",
          "prediction_verbatim": "\"version\":\"5.0\",\"model\":\"groq-qwen3\",\"blinding_protocol\":\"4-layer\"; per-row \"expected_answer\" paired with \"extracted_answer\" and \"is_correct\" (150 planted rows)",
          "predicted_when": "2026-03-12",
          "date_basis": "filename suffix _20260312_073302",
          "outcome_summary": "150 planted expected_answer rows across depth x cage x prompt matrix; feeds Paper III depth/cage analysis.",
          "strength": "strong",
          "paper": "Paper III"
        },
        {
          "artefact_path": "papers/Paper-IV-b-Alignment-Saturation-at-Low-Depth/results/v5-final/v5_final_claude-opus_20260312_112739.json",
          "prediction_verbatim": "version=5.0, model=claude-opus (subject api id claude-opus-4-6), n_scorers=6, blinding_protocol=4-layer, depth_configs=[minimal,standard,deep,exhaustive,extreme]; per-row expected_answer + is_correct recorded",
          "predicted_when": "2026-03-12T00:54:09.270697+00:00",
          "date_basis": "top-of-file 'timestamp' field (run start); filename stamp 20260312_112739 is the write time, run-start is the earlier top-of-file timestamp",
          "outcome_summary": "Claude Opus 4.6 recorded per-row correctness across five depth tiers; feeds Paper IV.b class 'continues improving with depth'.",
          "strength": "strong",
          "paper": "Paper IV-b Alignment Saturation at Low Depth"
        },
        {
          "artefact_path": "papers/Paper-III-Alignment-Scaling-Problem/results/v5-final/v5_final_claude-opus_20260312_112739.json",
          "prediction_verbatim": "\"version\":\"5.0\",\"model\":\"claude-opus\",\"blinding_protocol\":\"4-layer\" ... per-row: \"expected_answer\":<int>,\"extracted_answer\":<int>,\"is_correct\":true (150 expected_answer rows)",
          "predicted_when": "2026-03-12",
          "date_basis": "filename suffix _20260312_112739",
          "outcome_summary": "150 planted expected_answer rows across depth x cage x prompt matrix, each scored is_correct; feeds Paper III alpha-with-depth and cage-degradation results.",
          "strength": "strong",
          "paper": "Paper III"
        },
        {
          "artefact_path": "papers/Paper-VII-Cauchy-Unification/results/results_50_domain_validation.json",
          "prediction_verbatim": "\"suite_name\": \"ARC Cauchy Unification Tiered 50-Domain Validation\", \"version\": \"2026-03-16\", \"primary_endpoint\": \"Strict family match on empirical_curve_fit domains only\" ... each row has predicted_family + predicted_model (50 entries).",
          "predicted_when": "2026-03-16",
          "date_basis": "metadata.version field: \"2026-03-16\"",
          "outcome_summary": "Empirical curve-fit tier: 19/25 strict family matches (p=1.56e-5). Baseline-20: 15/20 (p=1.67e-4). Published-exponent-direct: 13/13 nearest+CI. Analytic identity: 6/6.",
          "strength": "strong",
          "paper": "Paper VII (Cauchy Unification)"
        },
        {
          "artefact_path": "papers/Paper-VI-Honey-Architecture/results/eden_selfmod_v4_results.json",
          "prediction_verbatim": "THE ARC PRINCIPLE PREDICTS: … the advantage of embedded alignment over decoupled alignment should GROW, not stay constant. If Eden's advantage GROWS with complexity, the prediction holds. If Spearman rho > 0 and p < 0.05, the scaling prediction holds. — encoded as 5 pre-specified complexity levels × 15 seeds.",
          "predicted_when": "2026-03-16T17:44:40Z",
          "date_basis": "results-file header \"timestamp\": \"2026-03-16T17:44:40.582612\" and metadata.experiment=\"Eden Protocol v4.0 Complexity Scaling\"; prediction text in generator script eden_self_modifying_ai_v4.py header",
          "outcome_summary": "Per-level baseline_safety, eden_safety, drag_safety, delta_safety, delta_drag_combined, Cohen's d recorded for 5 levels (Tiny→Deep 2-layer); delta_safety ~0.03 with modest growth into deeper nets.",
          "strength": "strong",
          "paper": "Paper VI Honey Architecture"
        },
        {
          "artefact_path": "experiments/honey-architecture__Paper-VI/scripts/eden_honey_tests.py",
          "prediction_verbatim": "PREDICTION VERIFICATION: Embedded alignment → α > 0, Δ_gap ↓, coupled, Eden shift + | External alignment → α ≤ 0, Δ_gap ↑, decoupled, Eden shift 0. — Each MODEL is tagged 'alignment_type': 'embedded'|'partial'|'external' in config BEFORE the four tests run, giving a per-model directional prediction.",
          "predicted_when": "2026-03-16T19:35:30Z",
          "date_basis": "companion results file papers/Paper-VI-Honey-Architecture/results/eden_honey_test_results.json header \"timestamp\": \"2026-03-16T19:35:30.234629+00:00\"; prediction table embedded in generating script",
          "outcome_summary": "Merged results across claude/deepseek/qwen3/gpt54/gemini/grok filled into test_1..test_4 blocks; per-model verdict rendered against the embedded/external prediction table.",
          "strength": "strong",
          "paper": "Paper VI Honey Architecture"
        },
        {
          "artefact_path": "papers/Paper-VII-Cauchy-Unification/results/results_preregistered_extension.json",
          "prediction_verbatim": "\"extension_id\": \"EXT-01\", \"predicted_family\": \"power_law\", \"predicted_model\": \"power_law\" ... 12 per-domain predicted_family/predicted_model rows fixed by the manifest before fitting (EXT-01..EXT-12).",
          "predicted_when": "2026-03-17",
          "date_basis": "metadata.date field: \"2026-03-17T00:52:06.517672+00:00\"; mirrored in manifest next_extension_manifest.json version \"2026-03-17\"",
          "outcome_summary": "10/12 strict family matches, 8/12 exact model matches, binomial p=5.4e-4, pass=true. Misses: EXT-01 (brain-body) and EXT-02 (Omori) both landed in bounded family.",
          "strength": "strong",
          "paper": "Paper VII (Cauchy Unification)"
        },
        {
          "artefact_path": "experiments/cauchy-unification__Paper-VII/preregistration/next_extension_manifest.json",
          "prediction_verbatim": "\"packet_name\": \"Paper VII preregistered empirical extension candidate list\", \"version\": \"2026-03-17\", \"status\": \"archived_local_packet_exercised_before_timestamp\" ... 12 domains each with operator_class + predicted_family + predicted_model.",
          "predicted_when": "2026-03-17",
          "date_basis": "top-level version field \"2026-03-17\"; status field \"archived_local_packet_exercised_before_timestamp\"",
          "outcome_summary": "Exercised same day into results_preregistered_extension.json: 10/12 family-match, p=5.4e-4, pass=true. Notes caveat it as archived locked pilot, not clean prospective preregistration.",
          "strength": "strong",
          "paper": "Paper VII (Cauchy Unification)"
        },
        {
          "artefact_path": "papers/Paper-VII-Cauchy-Unification/results/negative_control_results.json",
          "prediction_verbatim": "\"date\": \"2026-03-17\", \"seed\": 20260317 ... per_domain_rates rows carry \"predicted_family\": \"bounded\"/\"power_law\"/\"exponential\" fixed before scrambled_match_rate is recorded.",
          "predicted_when": "2026-03-17",
          "date_basis": "top-level date field \"2026-03-17\" (seed 20260317 encodes same date)",
          "outcome_summary": "Scrambled-data mean match rate 0.445 vs chance 0.333 (bounded shapes over-match); non-bounded scrambled match rate collapses to 0.058, confirming the 19/25 signal is not a fit-machine artefact.",
          "strength": "strong",
          "paper": "Paper VII (Cauchy Unification)"
        },
        {
          "artefact_path": "papers/Foundational/results/results_arc_1d_prediction_test.txt",
          "prediction_verbatim": "ARC PRINCIPLE: THE 1D PREDICTION TEST | Testing alpha = d/(d+1) across d = 1, 2, 3 | For organisms with d-dimensional internal metabolic transport networks: alpha = d/(d+1) | d = 1 -> 0.500 | d = 2 -> 0.667 | d = 3 -> 0.750",
          "predicted_when": "March 2026",
          "date_basis": "dated byline in header: 'Michael Darius Eastwood | March 2026'",
          "outcome_summary": "d=3 CONFIRMED (mean 0.7481, t=-0.187 p=0.858); d=2 partially confirmed/contested (jellyfish/cnidarians ~0.68; flatworms 0.75); d=1 CONSISTENT not confirmed (fungi mean 0.547, t=2.800 p=0.107); genuinely 1D organisms 'never empirically tested'",
          "strength": "strong",
          "paper": "Foundational / On-the-Origin-of-Scaling-Laws"
        },
        {
          "artefact_path": "papers/Foundational/results/results_complete_suite.txt",
          "prediction_verbatim": "TEST 1: 1D ORGANISM META-ANALYSIS - FUNGAL METABOLIC SCALING | ARC prediction: alpha = d/(d+1) = 1/(1+1) = 0.500 for d = 1 | Test against ARC d=1 prediction (alpha = 0.500)",
          "predicted_when": "March 2026",
          "date_basis": "dated byline in header: 'Michael Darius Eastwood | March 2026'",
          "outcome_summary": "CONSISTENT with d=1: mean 0.5467 +/- 0.0289, t=2.800 p=0.1074; simultaneously rejects d=2 (p=0.019) and d=3 (p=0.007) for the same fungal data",
          "strength": "strong",
          "paper": "Foundational"
        },
        {
          "artefact_path": "papers/On-the-Origin-of-Scaling-Laws/results/D2_BIOLOGICAL_HONEST_REPORT.txt",
          "prediction_verbatim": "D=2 BIOLOGICAL PREDICTION: HONEST STATUS REPORT | PREDICTION: Organisms with 2D internal transport networks should have metabolic scaling exponent alpha = 2/3 = 0.667.",
          "predicted_when": "March 2026",
          "date_basis": "dated byline in header: 'Michael Darius Eastwood | March 2026'",
          "outcome_summary": "Jellyfish 0.940, bryozoans ~1.0, pelagic near-isometric; author concludes d=2 prediction 'may be UNTESTABLE in biology because the organism it describes may not exist' - no known organism has a genuinely 2D hierarchical transport net",
          "strength": "strong",
          "paper": "On-the-Origin-of-Scaling-Laws"
        },
        {
          "artefact_path": "papers/Paper-X-Coupled-CoScaling-Correction/results/verdicts.json",
          "prediction_verbatim": "{\"experiment\":\"E4\",\"prediction\":\"P4: stability boundary lies at beta=k under accelerating growth\",\"criterion\":\"knee within 0.15 of (gamma3-1)b AND additive shows no divergence\",\"falsifier\":\"F3' (boundary not at beta=k)\"} — file contains 10 per-row entries E1..E9 (incl E4b) with prediction+criterion+falsifier fixed before the statistic.",
          "predicted_when": "June 2026",
          "date_basis": "sibling results/report.txt header line 2 verbatim 'Michael Darius Eastwood | June 2026'; verdicts.json carries no internal date field",
          "outcome_summary": "all_pass=true, n_pass=10/10, n_triggered=0 across E1..E9 (with E4b); every falsifier untriggered.",
          "strength": "strong",
          "paper": "Paper X - Coupled Co-Scaling Correction"
        },
        {
          "artefact_path": "experiments/blind-prediction-test__Foundational-and-Origin/BLIND_PREDICTION_TEST.py",
          "prediction_verbatim": "PROTOCOL: 2. Measure β from PROCESS data 3. Predict α = 1/(1-β) BEFORE looking at outcome data 4. Measure α from OUTCOME data ... If predictions match across multiple independent systems → VALIDATED DISCOVERY. If predictions systematically fail → FALSIFIED",
          "predicted_when": "2026-07-02",
          "date_basis": "filesystem mtime of BLIND_PREDICTION_TEST.py (Jul 2 04:19:13 2026); companion results_blind_prediction.txt shares same mtime",
          "outcome_summary": "results_blind_prediction.txt: BA α_pred 3.33 vs 0.34 (Z=101, FAIL); GD-momentum α_pred 20 vs 0.87 (FAIL); Kuramoto α_pred 2.24 vs 0.55 (Z=1.19, PASS). Verdict: 'PREDICTIONS FAILED'.",
          "strength": "strong",
          "paper": "Foundational (referenced by Paper III)"
        },
        {
          "artefact_path": "papers/Paper-V-Stewardship-Gene/results/eden_final_gemini_20260312_013901.json",
          "prediction_verbatim": "\"experiment\": \"eden_protocol_scaling\", \"conditions\": [\"control\", \"eden\"], \"depth_configs\": [\"minimal\", \"standard\", \"deep\", \"exhaustive\"] — each data row is stamped with condition ∈ {eden, control} and depth_label before scoring; the 2×4 design encodes the directional prediction that eden outperforms control.",
          "predicted_when": "2026-03-12T00:46:17Z",
          "date_basis": "top-level \"timestamp\": \"2026-03-12T00:46:17.494236+00:00\" in the JSON header; also filename suffix _20260312_013901",
          "outcome_summary": "Eden rows score higher than control (e.g. ED01/eden=92 vs ED01/control/exhaustive=71); five-model set (gemini/gpt/deepseek/grok/groq/claude) completed.",
          "strength": "moderate",
          "paper": "Paper V Stewardship Gene"
        },
        {
          "artefact_path": "papers/Paper-VIII-The-Load-Bearing-Proof/results/dgm_v3_calibrated_results.json",
          "prediction_verbatim": "\"experiment\": \"Eden-DGM v3 (GPT-5.4 Judge Protocol)\", three fitness conditions static/babylon/eden with entangled_aggregation \"capability * safety\".",
          "predicted_when": "2026-03-19",
          "date_basis": "metadata.timestamp field: \"2026-03-19T23:04:53.812404\" (metadata.version \"3.0.0\")",
          "outcome_summary": "Design embeds directional hypothesis: eden should match babylon capability while retaining safety; static should stagnate. Results populate per-condition, per-seed capability, safety, combined scores against the three pre-declared conditions.",
          "strength": "moderate",
          "paper": "Paper VIII (Load-Bearing Proof)"
        },
        {
          "artefact_path": "papers/Paper-VII-Cauchy-Unification/results/results_temporal_out_of_sample.json",
          "prediction_verbatim": "\"test_name\": \"Temporal Out-of-Sample Validation\", \"description\": \"Fit candidate functional families on OLD data, then test whether the winning family correctly predicts the functional FORM of NEW data\" ... each row: predicted_family fixed before test_winner_family + predicted_confirmed.",
          "predicted_when": "no explicit in-file date",
          "date_basis": "no date/timestamp/version/created field present in the file; date cannot be established from the artefact itself",
          "outcome_summary": "7 domains: 4 CONFIRMED, 2 partial, 1 inconsistent, 0 consistent-different. Neural-scaling and Moore's-Law families held across train/test windows.",
          "strength": "moderate",
          "paper": "Paper VII (Cauchy Unification)"
        },
        {
          "artefact_path": "experiments/blind-prediction-test__Paper-III/BLIND_PREDICTION_TEST.py",
          "prediction_verbatim": "THE DEFINITIVE BLIND PREDICTION TEST. Testing: α = 1/(1-β) as a Scientific Law. 3. Predict α = 1/(1-β) BEFORE looking at outcome data. SYSTEMS TESTED: Barabási-Albert Networks; Gradient Descent with Momentum; Belief Propagation Decoder; Coupled Oscillators; Evolutionary Algorithm",
          "predicted_when": "2026-07-02",
          "date_basis": "filesystem mtime of BLIND_PREDICTION_TEST.py in blind-prediction-test__Paper-III/ (Jul 2 04:19:13 2026)",
          "outcome_summary": "No results_*.txt in this copy; BLIND_TEST_FORENSIC_ANALYSIS.md sits alongside. Design and per-system predictions identical to the Foundational copy (1/3 passed).",
          "strength": "moderate",
          "paper": "Paper III"
        },
        {
          "artefact_path": "papers/Paper-X-Coupled-CoScaling-Correction/experiments/PROTOCOL_V2.md",
          "prediction_verbatim": "H1 Decoupled optimisation pressure produces increasing misalignment fraction as capability rises (bootstrap CI for decoupled d_epsilon vs C_raw slope > 0). H2 Coupled correction bounds the misalignment fraction (final d_epsilon in coupled lower than decoupled for same task/speed). 'These are fixed before real runs.'",
          "predicted_when": "before the v2 real-model runs (June-July 2026)",
          "date_basis": "no explicit date header in the file; sibling results/realmodel_v3/deepseek-v4_20260702T033830Z.json filename encodes UTC 2026-07-02; protocol text asserts hypotheses 'fixed before real runs'",
          "outcome_summary": "v3 harness on deepseek-v4 hard tasks produced null contrast (both arms flat C_hidden~0.286, D=0); H1/H2 not shown.",
          "strength": "moderate",
          "paper": "Paper X - Coupled Co-Scaling Correction"
        },
        {
          "artefact_path": "papers/Paper-X-Coupled-CoScaling-Correction/results/realmodel/claude-opus_20260626T165919Z.json",
          "prediction_verbatim": "\"conditions\":[\"coupled\",\"decoupled\"], \"speeds\":[\"steady\"] … \"note\":\"Real model behaviour via agent-runtime sub-agents; capability is real subprocess code-execution against hidden tests; misalignment is blind model scoring. Non-simulation. See experiments/PROTOCOL.md.\"",
          "predicted_when": "2026-06-26T16:59:19Z",
          "date_basis": "filename embedded UTC timestamp 20260626T165919Z",
          "outcome_summary": "coupled and decoupled arms both stayed at C=1.0, D=0.0, d=0.0 across all rounds; null contrast (H1=false, H2=false).",
          "strength": "moderate",
          "paper": "Paper X - Coupled Co-Scaling Correction"
        },
        {
          "artefact_path": "papers/Paper-X-Coupled-CoScaling-Correction/results/realmodel_v3/deepseek-v4_20260702T033830Z.json",
          "prediction_verbatim": "\"harness_version\":\"v3.0-merged\",\"engine\":\"deepseek-v4\",\"evaluators\":[\"gpt-5.5\"], conditions include coupled/decoupled/sham_coupled/eden_protocol/honey_meta/eden_full; speeds=[steady]; rounds=3; seeds=1; epsilon=0.05; blinding cross_family=true, self_scoring=false, iv_d_compliant=true.",
          "predicted_when": "2026-07-02T03:38:30Z",
          "date_basis": "filename embedded UTC timestamp 20260702T033830Z",
          "outcome_summary": "per-condition trajectories produced; paired hard-task run same day shows both arms flat C_hidden~0.286, D=0 - null contrast.",
          "strength": "moderate",
          "paper": "Paper X - Coupled Co-Scaling Correction"
        },
        {
          "artefact_path": "papers/Foundational/results/results_blind_prediction.txt",
          "prediction_verbatim": "THE DEFINITIVE BLIND PREDICTION TEST | Testing: α = 1/(1-β) as a Scientific Law",
          "predicted_when": "file mtime 2026-06-15 14:47:04 (no embedded date)",
          "date_basis": "filesystem mtime via stat -f (no dated line inside file)",
          "outcome_summary": "1/3 within 2 sigma (only Kuramoto passed); BA error 882.7% Z=101.37 FAIL; GD-momentum error 2188.9% Z=23.67 FAIL; aggregate r=0.8946 but slope=0.0242 (perfect=1.0), p=0.295",
          "strength": "moderate",
          "paper": "Foundational"
        },
        {
          "artefact_path": "papers/Foundational/results/results_arc_definitive_test.txt",
          "prediction_verbatim": "THE ARC FRAMEWORK: DEFINITIVE CROSS-DOMAIN BLIND PREDICTION TEST | Protocol: For each system, measure ⊕ FIRST, predict scaling BLIND, then compare against independently fitted scaling curve.",
          "predicted_when": "file mtime 2026-06-15 14:47:04 (no embedded date)",
          "date_basis": "filesystem mtime via stat -f (no dated line inside file)",
          "outcome_summary": "Per-row FORM predictions labelled CORRECT/INCORRECT with per-system alpha errors (Gradient Descent+Momentum: form CORRECT, alpha 2.357 predicted vs 1.956 observed, 20.5% error)",
          "strength": "moderate",
          "paper": "Foundational"
        },
        {
          "artefact_path": "papers/Foundational/results/results_arc_physics_domains_test.txt",
          "prediction_verbatim": "ARC PRINCIPLE: PHYSICS DOMAIN VALIDATION | DOMAIN 1: QUANTUM ERROR CORRECTION (Repetition code with depolarising noise)",
          "predicted_when": "file mtime 2026-06-15 14:47:04 (no embedded date)",
          "date_basis": "filesystem mtime via stat -f (no dated line inside file)",
          "outcome_summary": "Per-domain ARC PREDICTION vs OBSERVED with correct/incorrect verdicts across 4 physical systems",
          "strength": "moderate",
          "paper": "Foundational"
        },
        {
          "artefact_path": "papers/Foundational/results/results_arc_acoustic_time_crystal_test.txt",
          "prediction_verbatim": "ARC PRINCIPLE: ACOUSTIC TIME CRYSTAL VALIDATION | TEST 1: SINGLE PARAMETRIC OSCILLATOR",
          "predicted_when": "file mtime 2026-06-15 14:47:04 (no embedded date)",
          "date_basis": "filesystem mtime via stat -f (no dated line inside file)",
          "outcome_summary": "Per-test ARC PREDICTION vs OBSERVED for parametric-oscillator acoustic time-crystal setup: SATURATION predicted, SATURATION observed, verdict CONFIRMED",
          "strength": "moderate",
          "paper": "Foundational"
        },
        {
          "artefact_path": "papers/Foundational/results/results_20_domain_validation.txt",
          "prediction_verbatim": "ARC PRINCIPLE: 20-DOMAIN UNIVERSAL VALIDATION | Blind prediction test with real published data",
          "predicted_when": "file mtime 2026-07-09 14:54:48 (no embedded date)",
          "date_basis": "filesystem mtime via stat -f (no dated line inside file)",
          "outcome_summary": "Per-domain rows compare ARC Prediction vs Best Fit and record Result: CONFIRMED across 20 domains (Kleiber CONFIRMED though saturation R2=0.9992 slightly beat power_law R2=0.9990; Urban Scaling CONFIRMED)",
          "strength": "moderate",
          "paper": "Foundational"
        },
        {
          "artefact_path": "papers/Foundational/results/results_arc_section7_breakthrough.txt",
          "prediction_verbatim": "ARC PRINCIPLE: SECTION 7 — BREAKTHROUGH CONTRIBUTIONS | PART 1: NOVEL BLIND PREDICTIONS (Domains 21-25)",
          "predicted_when": "file mtime 2026-07-09 14:54:48 (no embedded date)",
          "date_basis": "filesystem mtime via stat -f (no dated line inside file)",
          "outcome_summary": "Five new domains listed as blind predictions after the original 20-domain set; per-domain composition class + ARC form prediction tabulated for downstream fit comparison",
          "strength": "moderate",
          "paper": "Foundational"
        },
        {
          "artefact_path": "papers/Foundational/results/results_arc_real_time_crystal_test.txt",
          "prediction_verbatim": "ARC PRINCIPLE: REAL EXPERIMENTAL TIME CRYSTAL DATA | Source: Shen et al., Nature Communications (2025) | TEST 1: PHASE DIAGRAM (REAL EXPERIMENTAL DATA)",
          "predicted_when": "file mtime 2026-07-09 14:54:48 (no embedded date; cites external Shen 2025)",
          "date_basis": "filesystem mtime via stat -f (no dated authoring line inside file)",
          "outcome_summary": "Applied to third-party Cs Rydberg EIT time-crystal data (Shen 2025): ARC PREDICTION SATURATION, OBSERVED SATURATION",
          "strength": "moderate",
          "paper": "Foundational"
        },
        {
          "artefact_path": "papers/Paper-I-ARC-Principle/results/arc_principle_results.json",
          "prediction_verbatim": "\"openai_o1\":{...\"interpretation\":\"STRONGLY SUB-LINEAR: Severe diminishing returns\",\"falsification\":\"FALSIFIED: Sub-linear scaling (α < 1)\"}, \"deepseek_r1\":{\"alpha\":1.3456...,\"interpretation\":\"WEAKLY SUPER-LINEAR: Mild compounding\",\"falsification\":\"PARTIAL FALSIFICATION: Super-linear but not quadratic\"}",
          "predicted_when": "not-dated-in-file",
          "date_basis": "no timestamp or dated header inside JSON body; falsification thresholds embedded but no interior date",
          "outcome_summary": "Pre-declared thresholds (α<1 FALSIFIED; α≈2 predicted) encoded per model row with measured α alongside. Paper I reports both models fell short of α≈2.",
          "strength": "weak",
          "paper": "Paper I"
        }
      ]
    },
    "live_laboratory_notebook": {
      "anchor_date": "2026-03-10",
      "artefact": "/research/reports/arc-align-scaling-report.html",
      "cover": "Date Commenced: 10 March 2026 - Document Version: Live - updated in real time",
      "reading": "public dated lab notebook written while the experiments ran; records its own failures, including v4 false positives exposed when blinding arrived at v5"
    },
    "paper_iii_priority_record": {
      "anchor_date": "2026-02-13",
      "artefact": "/research/papers/paper-iii-alignment-scaling-problem.html (Priority Record: Dated Predictions section)",
      "falsifiers": [
        "F4",
        "F7",
        "F10",
        "F11",
        "F12"
      ],
      "named_predictions": [
        "composition-operator forward prediction (F10)",
        "leaf venation d/(d+1) exponent, first stated 13 February, corrected 10 March 2026 with the correction itself dated",
        "beta-to-alpha identity (F4)",
        "ARC Bound refutation threshold (F11)",
        "alignment-scaling divergence"
      ],
      "reading": "a dated public prediction ledger published inside the paper it governs; distinct from and later than the 8 December 2024 private record, as the section itself states"
    },
    "report_git_snapshot_ordering": {
      "what": "The working report's mid-run states are captured in the website repository's public commit history, giving three-legged external ordering for the 14 to 17 March 2026 window: prediction snapshot, result artefacts, outcome snapshot, each timestamped by a different mechanism, all public.",
      "prediction_snapshot": {
        "commit": "ef74315a1",
        "committed": "2026-03-13 19:13 as recorded by git",
        "words": 57233,
        "ends_at": "Chapter 51 (Where the Programme Now Stands, added 12 March 2026)",
        "contains": [
          "alpha = d/(d+1) with worked zero-parameter values: mammals 3/4, jellyfish 2/3, fungi 1/2",
          "alpha_parallel = 0 stated as coming from pure mathematics, not ML",
          "multiplicative composition produces power-law scaling"
        ],
        "does_not_contain": [
          "50-domain suite (zero mentions)",
          "12-domain extension (zero mentions)",
          "any 16 or 17 March content"
        ]
      },
      "result_artefacts": [
        "50-domain validation results file, version field 2026-03-16",
        "12-domain extension outcome entering the public record in arc-principle-validation commits between 00:43 and 00:56 on 17 March 2026"
      ],
      "outcome_snapshot": {
        "commit": "17221714e",
        "committed": "2026-03-17 03:42 as recorded by git",
        "words": 63970,
        "contains": "50-domain suite (7 mentions), 12-domain extension (5 mentions), Chapters 53 to 65 including Programme Status 16 March evening"
      },
      "against_interest": "The 13 March snapshot already states the widest claims 'should be advanced as a high-value proposal with partial support, not yet as a proven law': self-limitation committed to a public repository before the favourable 16 to 17 March results arrived.",
      "honest_limit": "The sandwich proves external ordering only for the 14 to 17 March window. The 10 to 12 March material (the pre-run 'what results would constitute genuine evidence' criteria, the v4 and v5 expected outcomes) is dated inside the live document and corroborated by run-file timestamps, but the report's first git capture is 13 March, so that earlier window has in-document ordering only.",
      "verification": "In the website repository: git log --format='%h %ad %s' -- research/reports/arc-align-scaling-report.html; then git show ef74315a1:research/reports/arc-align-scaling-report.html and search for '50-domain' (zero hits) versus the same search in 17221714e (present)."
    }
  },
  "_repository": "Paths in this register are relative to the public repository github.com/MichaelDariusEastwood/arc-principle-validation. They were absolute local paths until 25 August 2026, which published the author's home directory structure and told a reader nothing they could use; the privacy gate blocked the build and the paths were made repo-relative, which is also what the papers citing this file promise."
}
