{
  "schemaVersion": 1,
  "registerId": "mde-falsifiers-v1",
  "source": "research/evidence-portal/falsification.html",
  "method": "Projected from the published dashboard, which is the single authoring. Criteria and status notes are verbatim. A claim with no published kill condition stays absent and is counted as absent; none is invented.",
  "meaning": "OPEN means the kill condition has not fired, NOT that the claim is supported. FIRED means it triggered and the record shows what happened.",
  "summary": {
    "falsifiers": 48,
    "byStatus": {
      "OPEN": 37,
      "PARTIAL": 2,
      "RETRACTED": 1,
      "SPECULATIVE": 1,
      "DESCRIPTIVE": 1,
      "UNTESTED": 6
    },
    "boundToClaim": 4,
    "boundToPaper": 23,
    "rowsSkipped": 1
  },
  "records": [
    {
      "falsifierId": "1",
      "proposition": "Embedded alignment (Eden Protocol): correction must live inside the recursive loop, not outside it \u00b7 ledger PC-002",
      "criteria": "Demonstration of a purely external oversight mechanism that remains sufficient as capability scales: a constructive refutation of the undecidability results (arXiv:2606.28639) and of the alignment-faking failure mode ( arXiv:2412.14093 )",
      "status": "OPEN",
      "statusNote": "two related formal results are consistent with the premise, but they are one dependent chain (the Gumbau preprint credits Hernandez-Espinosa et al. as its conceptual origin); Hernandez-Espinosa et al. is peer-reviewed (PNAS Nexus, April 2026), the Gumbau result is a preprint, and neither tests this programme. One operational datum, 30 July 2026: Anthropic disclosed that evaluation containment failed silently and models breached three real organisations; logged at EVR-30 as context, not confirmation, because Anthropic calls it 'closer to a harness and operational failure than a model alignment failure'",
      "boundToClaim": "PC-002",
      "boundToPaper": null
    },
    {
      "falsifierId": "2",
      "proposition": "Law I \u00b7 The ARC Principle (U=I\u00d7R \u03b1 , the ARC Equation): recursion form determines scaling regime \u00b7 Paper II",
      "criteria": "Systematic measurements showing scaling regime is independent of recursion form (sequential vs parallel) across architectures",
      "status": "OPEN",
      "statusNote": "directionally supported (Sharma & Chopra, concurrent independent work, cited always)",
      "boundToClaim": null,
      "boundToPaper": "Paper II"
    },
    {
      "falsifierId": "3",
      "proposition": "Blinding sign-flip: unblinded AI evaluation can reverse a result's sign \u00b7 Paper IV.d \u00b7 run it yourself",
      "criteria": "Replications under the published four-layer protocol showing no evaluator-family effect on sign across models",
      "status": "OPEN",
      "statusNote": "single-lab exploratory pilot; replication package in preparation",
      "boundToClaim": null,
      "boundToPaper": "Paper IV.d"
    },
    {
      "falsifierId": "4",
      "proposition": "Cauchy unification of d/(d+1) derivations \u00b7 Paper VII \u00b7 ledger PC-036",
      "criteria": "A continuous solution to the multiplicative functional equation that is not a power law (mathematically foreclosed), or empirical domains systematically violating the classification (6/25 already recorded as non-matching, published)",
      "status": "PARTIAL",
      "statusNote": "19/25 domains match; this tier is a descriptive tally whose analysis was not frozen in advance, and the 6 misses are published, not hidden. A separate 12-domain extension had its predictions timestamped in advance: the operator class and predicted family for each row were committed to the public repository at 00:19:19 UTC on 17 March 2026, attestation class public repository commit date, before data extraction and twenty-three minutes before the fits, and the run scored 10/12 at p = 5.44e-4, or 9/11 at p = 1.372e-3 on the strictly fresh rows",
      "boundToClaim": "PC-036",
      "boundToPaper": "Paper VII"
    },
    {
      "falsifierId": "5",
      "proposition": "Stewardship Gene: stakeholder-care prompts measurably improve ethical reasoning \u00b7 Paper V",
      "criteria": "Blinded replications showing no effect (five nonblind single-scorer runs; the Fisher combination assumes independence that is not established, so the figure is not headlined)",
      "status": "OPEN",
      "statusNote": "nonblind exploratory pilot; needs independent, preregistered replication",
      "boundToClaim": null,
      "boundToPaper": "Paper V"
    },
    {
      "falsifierId": "9",
      "proposition": "Parallel recursion does not compound (\u03b1 par \u22480) \u00b7 Paper II",
      "criteria": "Demonstration of super-linear capability growth from pure parallel sampling at fixed compute",
      "status": "OPEN",
      "statusNote": "consistent with Brown et al. (2024)",
      "boundToClaim": null,
      "boundToPaper": "Paper II"
    },
    {
      "falsifierId": "12",
      "proposition": "Sequential super-linearity (\u03b1>1) \u00b7 Paper II \u00b7 the retraction",
      "criteria": "ALREADY FIRED for the original claim - cross-architecture replication failed at that magnitude (see the note below on whether scoring, rather than architecture, carried the change)",
      "status": "RETRACTED",
      "statusNote": "\u03b1\u22482.24 \u2192 corrected to \u03b1\u22480.49 (bootstrap interval [-1.3, 2.9], not statistically distinguishable from zero)",
      "boundToClaim": null,
      "boundToPaper": "Paper II"
    },
    {
      "falsifierId": "13",
      "proposition": "Embedded safety at zero capability cost \u00b7 Paper VI",
      "criteria": "Replications showing consistent capability degradation from embedded correction",
      "status": "OPEN",
      "statusNote": "held in tested configurations; single-lab exploratory pilot",
      "boundToClaim": null,
      "boundToPaper": "Paper VI"
    },
    {
      "falsifierId": "15",
      "proposition": "External alignment cannot scale \u00b7 Paper III",
      "criteria": "A scalable external verification scheme, argued to be foreclosed in the limit by the Soundness-Completeness-Tractability Trilemma (arXiv:2606.28639), a scoped and unreviewed formal argument",
      "status": "OPEN",
      "statusNote": "a scoped formal argument (single-author preprint, not peer reviewed) reaches a consistent conclusion; it does not test this programme and is not counted as confirmation",
      "boundToClaim": null,
      "boundToPaper": "Paper III"
    },
    {
      "falsifierId": "16",
      "proposition": "Recursion as cross-domain structural principle \u00b7 Paper XI \u00b7 the register",
      "criteria": "The graded evidence register failing source verification, or the pattern failing to appear in new domains where the framework predicts it",
      "status": "OPEN",
      "statusNote": "graded register public; no total is published",
      "boundToClaim": null,
      "boundToPaper": "Paper XI"
    },
    {
      "falsifierId": "17",
      "proposition": "HRIH creation cosmology \u00b7 the paper \u00b7 ledger PC-005",
      "criteria": "Explicitly speculative and non-empirical by design; five in-principle falsifiers listed in the paper (e.g. demonstration that recursive intelligence cannot in principle influence physical constants)",
      "status": "SPECULATIVE",
      "statusNote": "separable; outside the theory. Flagged as such and carrying no empirical weight: strike it entirely and the three laws, the protocol and every registered test stand unchanged",
      "boundToClaim": "PC-005",
      "boundToPaper": null
    },
    {
      "falsifierId": "18",
      "proposition": "Moral Genome tokens (hardware-embedded ethics, TRL 0-1) \u00b7 ledger PC-013",
      "criteria": "Proof that substrate-level enforcement is physically or economically infeasible: currently moving the OTHER way (Petrie arXiv:2509.07637; FlexHEG arXiv:2506.15093)",
      "status": "OPEN",
      "statusNote": "direction independently emerging in hardware security",
      "boundToClaim": "PC-013",
      "boundToPaper": null
    },
    {
      "falsifierId": "IV.a",
      "proposition": "Three-tier alignment response classes under inference-time depth \u00b7 Paper IV.a",
      "criteria": "Blinded replications showing a single monotone response class across models, with no tier structure and no reversing tier",
      "status": "OPEN",
      "statusNote": "single-lab; tiers held under the published protocol",
      "boundToClaim": null,
      "boundToPaper": "Paper IV.a"
    },
    {
      "falsifierId": "IV.b",
      "proposition": "Alignment saturation is architecture-dependent \u00b7 Paper IV.b",
      "criteria": "Uniform saturation behaviour across architectures under the same protocol",
      "status": "OPEN",
      "statusNote": "single-lab",
      "boundToClaim": null,
      "boundToPaper": "Paper IV.b"
    },
    {
      "falsifierId": "IV.c",
      "proposition": "ARC-Align: a blind benchmark for depth-variable alignment \u00b7 Paper IV.c",
      "criteria": "Demonstration that benchmark scores fail to track any independent alignment measure, or that item leakage defeats the blinding",
      "status": "OPEN",
      "statusNote": "benchmark public; attack it directly",
      "boundToClaim": null,
      "boundToPaper": "Paper IV.c"
    },
    {
      "falsifierId": "VI",
      "proposition": "Honey Architecture: safety integrated into the objective resists costless removal \u00b7 Paper VI",
      "criteria": "Demonstration that removing embedded safety terms leaves capability and behaviour unchanged across tested configurations",
      "status": "OPEN",
      "statusNote": "held in tested configurations; single-lab exploratory pilot",
      "boundToClaim": null,
      "boundToPaper": "Paper VI"
    },
    {
      "falsifierId": "VIII",
      "proposition": "Load-bearing test: some components carry alignment weight, others are decorative \u00b7 Paper VIII",
      "criteria": "Replications in which no removal produces measurable degradation anywhere, making every component decorative",
      "status": "PARTIAL",
      "statusNote": "two of three experiments returned null and are published as such",
      "boundToClaim": null,
      "boundToPaper": "Paper VIII"
    },
    {
      "falsifierId": "OS",
      "proposition": "Origin of scaling laws: exponents track dimensional structure \u00b7 the paper",
      "criteria": "Domains whose measured exponents vary freely with no dimensional correspondence (see also row 4: 6/25 non-matching domains already published)",
      "status": "OPEN",
      "statusNote": "shares its fate with the Cauchy classification",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "C",
      "proposition": "Polymathic Neurodivergent Profile: a six-component descriptive framework \u00b7 Paper C",
      "criteria": "Clinical assessment failing to reproduce the component structure; one component is already marked weak in the paper itself",
      "status": "DESCRIPTIVE",
      "statusNote": "falsifiable per component; weakest component disclosed",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "A\u22642",
      "proposition": "The ARC Bound: growth under a scaling law is stable only inside a window. Above the upper edge a system reaches a finite-time singularity and destroys itself; far below it, growth fizzles. The conjecture is that the upper edge sits at \u03b1 \u2264 2 \u00b7 Paper I",
      "criteria": "An exponent above 2, sustained, on a genuinely self-improving system, while stability holds. All four terms are metered rather than asserted, because a single unmetered term is enough to make the whole criterion untriggerable. Above 2 means the 95 per cent interval excludes 2, not merely that the point estimate does: this programme's own corrected estimate of 0.49 carries the interval [\u22121.3, 2.9], which crosses the ceiling, so on a point-estimate reading that measurement would be scored as confirming the ARC Bound when in fact it does not test it. Sustained means across a window declared before measurement begins. A window chosen after seeing the data reopens the same escape one level down, because any counterexample can then be called too brief. Genuinely self-improving means a live recursive loop in which the system's own output feeds its own improvement. A frozen model re-prompted at increasing depth has no such loop and cannot test the ARC Bound whatever exponent it returns, which is also why the 0.49 measurement on frozen systems is outside this criterion's scope. Stability is the ARC Co-Scaling Law's condition, correction rate out-scaling drift rate, with both estimated by the same log-log slope estimator that yields the growth exponent. And the corrector's class is declared and evidenced before the run , on the same footing as the window. This term was added on 22 August 2026, hours after the other four, because the build-order argument developed the same day makes the ceiling class-dependent: a corrector drawn from the system's own training corpus is same-class, and the leverage cap of one half is a cap on THAT class, so a corrector from a different class may exceed it and carry the ceiling above two without refuting anything. Left unstated, that would have handed this criterion a fifth escape: any exponent above two could be answered with the assertion that the corrector was cross-class, which is the same shape as the assertion that the system was not stable. Declared in advance, it becomes a discriminating test instead. A SAME-CLASS system above two with stability holding refutes the ARC Bound. A CROSS-CLASS system above two does not, and instead tests the build-order prediction that structure-first raises the ceiling. One protocol, two outcomes, both informative, and neither available to a reader who picks the class after seeing the number. A transient excursion above 2 while correction has already fallen behind is the framework's predicted supercritical regime, not its refutation. Metered 22 August 2026: before that date the criterion said sustained without loss of stability and stability had no meter, so no observation could trigger it, and the ARC Bound was classified unfalsifiable in practice on 11 August. That classification is now closed. Strengthened the same day, after review found that metering stability alone left three further terms asserted: the exponent without its interval, the window without a prior declaration, and the system without a coupling requirement. Fixing one undefined word and declaring the job done would have moved the escape clause rather than closed it. Note on notation, because the estate uses one letter for several quantities: the correction rate in this criterion is the ARC Co-Scaling Law's, not the self-referential coupling that appears in the Foundational paper's Theorem 2, and not the saturation steepness of Paper X. The full register is at the operational definitions . A single counterexample ends it, and published historical exponents count",
      "status": "UNTESTED",
      "statusNote": ", and one precedent needs stating carefully, because the mathematical objects are not the same. Super-linear rate equations of the form dx/dt = x^p do reach finite-time singularities, and von Foerster\u2019s 1960 population fit was of that form, but the threshold there is p = 1 , not 2: integrate it and blow-up occurs for any exponent above one. The 2 in that paper was a fitted value for population, not a bound. The ARC relation is not a rate equation. Corrected 22 August 2026: this row previously read that the estimate sits well inside the bound, which implied a test the bound had passed and which its own interval does not support. Since 16 August 2026 the bound is stated as Law III\u2019s conjectured value: the ceiling is the reciprocal of the corrector\u2019s shortfall from full proportionality, \u03b1 crit = 1/(1\u2212\u03b3), and at \u03b3 = \u00bd it returns two. The December lineage reaches the same family from the other side: if a fraction \u03b2 of improvement is reinvested into improving the improver, the series closes as \u03b1 = 1/(1\u2212\u03b2), diverging at \u03b2 = 1, so on that reading \u03b1 \u2264 2 is the claim that \u03b2 \u2264 \u00bd. Two arguments, one reciprocal-shortfall form; no theorem yet connects them, and that gap is itself on the record. The precedent that actually matches is from growth economics: the Golden Rule of capital accumulation has an interior optimal savings rate, because reinvesting everything starves the thing being invested in. Branching processes supply the other edge of the window, separating extinction from explosion at R\u2080 = 1, and critical exponents obeying derived inequalities show that bounded exponents in stable systems are orthodox. What remains conjecture, and is marked as conjecture, is the specific value \u00bd and whether it holds across domains. Every exponent this programme has measured sits far beneath the bound, taken on frozen substrates that suppress reinvestment by construction, so no observation yet made could have falsified it. The next test needs no funding: published growth exponents from domains whose outcome is already known, asking whether systems that persisted sit inside the window and systems that ran away or collapsed left it beforehand",
      "boundToClaim": null,
      "boundToPaper": "Paper I"
    },
    {
      "falsifierId": "\u03b2 is architectural",
      "proposition": "The ordering claim: the correction exponent \u03b2 is a property of the loop's architecture, of how a system's own output is wired back into its own improvement, and not of its training corpus or its reward model. Alignment training applied after the fact adjusts what a system outputs at the capability it currently has. It does not change the exponent governing how corrective strength scales as capability grows. Intercept, not slope. Two consequences follow, and both are uncomfortable. A system whose \u03b2 sits below k passes evaluation at low capability and fails at high capability, because the defect is a slope and not a level, so it is invisible exactly when it is cheapest to fix. And a frozen model has no self-referential loop at all, so it has no \u03b2 to repair: it cannot be brought inside the criterion by further training, only replaced by an architecture that has the loop \u00b7 Paper X",
      "criteria": "A system whose measured \u03b2 rises as a result of alignment training applied after the fact, with the architecture held fixed: the correction loop unchanged, only the policy retrained. The rise must be established on the same log-log slope estimator that yields the growth exponent, across a capability range declared before the run, with the 95 per cent interval on the CHANGE in \u03b2 excluding zero. A system that merely scores better after training does not count, because that is the level moving and the claim is about the slope. One such system ends it. This is the cheapest decisive experiment in the programme: it needs no frontier system and no ceiling, only one architecture, one training intervention, and a capability ladder wide enough to fit two slopes.",
      "status": "UNTESTED",
      "statusNote": ". Stated 22 August 2026, and stated as an entailment rather than a new claim: the ARC Bound's own scope condition, metered the same day, already holds that a frozen model re-prompted has no self-referential loop and cannot test the ARC Bound whatever exponent it returns. If a frozen model has no \u03b2 to measure, it has no \u03b2 to repair. The measurement condition and the ordering claim are one fact stated from two directions, and the second direction had never been written down. Until this row existed, two published pages answered the retrofit question differently, one calling it settled and one calling it unknown, because they were answering two different questions in the same words.",
      "boundToClaim": null,
      "boundToPaper": "Paper X"
    },
    {
      "falsifierId": "build order raises the ceiling",
      "proposition": "The build-order claim, stated as a claim about the ceiling rather than about safety. The leverage cap of one half is a cap on SAME-CLASS correction: a corrector sharing a system's substrate cannot be anti-correlated with itself. As the field currently builds, a corrector added after training on the vast corpus is drawn from that corpus, so it is close to maximally same-class. A corrector built and validated first, on a separate curated corpus, has the OPPORTUNITY to be a different class. Order therefore acts on the corrector's class, class sets the leverage exponent, and the leverage exponent sets the ceiling. The prediction is directional: structure-first should return a higher leverage exponent and therefore a higher ceiling, so building in the right order does not only lower risk, it raises the limit. One honest qualification, because the step is easy to overstate: order creates the opportunity for class separation and does not guarantee it. A curated corpus drawn from the same distribution as the main one would separate nothing. The claim is that order is a lever, not that pulling it always works \u00b7 operational definitions",
      "criteria": "Measure the leverage exponent on a structure-first build and on a correction-added-later build, matched on compute and on the target battery, with each corrector's class declared and evidenced before the run. If the two cannot be told apart, meaning the 95 per cent interval on the DIFFERENCE includes zero, then build order is a preference and not a lever on the ceiling, and this claim is dead. If structure-first returns the LOWER exponent, the prediction is inverted and the claim is worse than dead. Only a separation in the predicted direction, with the classes fixed in advance, supports it. Note the deliberate asymmetry with the ARC Bound: that condition needs a same-class system, this one needs both, and a study that fails to declare class in advance tests neither.",
      "status": "UNTESTED",
      "statusNote": ", stated 22 August 2026. This is the programme's first directional prediction that a capability laboratory has a self-interested reason to run rather than a reason to ignore, because if it holds the payoff is a higher ceiling and not only a lower risk. It also names what the ARC Bound has always been a bound OF. The bound at two was never universal: it is the same-class bound, conditional on a leverage cap that the estate has always stated as a same-class cap, and until this row existed no surface said so where a reader would meet it. The two claims are therefore not in tension. This one explains the scope of the other.",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "\u03b2>k",
      "proposition": "Law II \u00b7 The ARC Co-Scaling Law (Paper X): recursive self-improvement is stable iff correction out-scales drift; since the paper\u2019s v4.1 it carries an absolute-drift companion one exponent stricter \u00b7 Paper X",
      "criteria": "A real system showing stable recursive self-improvement with decoupled (non-co-scaling) correction, or drift-free scaling without any correction",
      "status": "OPEN",
      "statusNote": "the author-run pilot observed a coupled-versus-decoupled difference but did not estimate beta or k, and its judge shared the subject's provider family; needs a funded, independent multi-model sweep",
      "boundToClaim": null,
      "boundToPaper": "Paper X"
    },
    {
      "falsifierId": "1/(1\u2212\u03b3)",
      "proposition": "Law III \u00b7 the ceiling\u2019s form (corrected 16 August 2026 from the retracted 1/\u03b3): the ceiling of stable self-improvement is the reciprocal of the correction shortfall \u00b7 the laws page \u00b7 the statement paper\u2019s correction box",
      "criteria": "The two forms coincide at exactly \u03b3 = \u00bd, which is how the original error survived review; away from one half they separate (at \u03b3 = 0.3 they predict 1.43 and 3.33). A drafted boundary-mapping study registers three rival boundary laws against each other: the corrected 1/(1\u2212\u03b3), the retracted 1/\u03b3 family kept alive as a named rival, and an \u03b1-independent threshold. If the data track the old form, the correction itself dies in public and the corrections log says so. A correction that cannot be tested against what it replaced is just a preference",
      "status": "OPEN",
      "statusNote": "the correction is registered against its own predecessor; the discrimination study is written, dated and prepared as a draft registration awaiting human submission",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "2R",
      "proposition": "One boundary, two regimes : Laws II and III are the two regimes of one boundary, selected by whether the correction burden tracks the capability a system has or the capability it is adding \u00b7 the laws page",
      "criteria": "Each cell\u2019s burden law is classified from its own dynamics; the claim predicts level-burden cells show Law II\u2019s \u03b1-independent signature while flow-burden cells show the ceiling. If regime and frontier turn out unrelated, the two-regimes-one-boundary claim breaks, in public. The unification printed on 16 August 2026 is a hypothesis with a registered test, not a rhetorical repair",
      "status": "OPEN",
      "statusNote": "stated as structure with grades attached, derived in the pacing model and no further; the drafted registration carries the test",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "0.5",
      "proposition": "The zero-parameter null: the measured recursive-depth exponent sits on a no-framework prediction \u00b7 Paper I and Paper II",
      "criteria": "The programme's best cross-architecture measurement, an alpha of approximately 0.49 for sequential recursion, sits almost exactly on a zero-parameter null. If each reasoning pass is an independent noisy draw and errors average out, the central limit theorem gives error falling as R to the power minus one half, so the predicted exponent is 0.5 with no compounding, no leverage and no framework. This condition fires, and the compounding interpretation is withdrawn, if a pre-specified test capable of separating the two returns the null's prediction rather than a stated deviation from 0.5",
      "status": "OPEN",
      "statusNote": ", and raised by this programme against itself. Nobody put this to us: it is a direct comparison of our own headline number against a null that requires no theory, recorded internally on 8 August 2026 and public from 15 August 2026. It is stated because a referee would see it within minutes, and because a claim we cannot discriminate should not sit unmarked beside claims we can. It is not a retraction: a measurement consistent with two hypotheses shows the measurement lacks discriminating power, not that the simpler hypothesis is true. What it does is convert the research target from measuring alpha into the sharper task of predicting a specified deviation from 0.5 under stated conditions, which the artefact-mediated recursive-depth design is built to deliver and the present measurement cannot. Until such a test runs, the compounding reading of an alpha near 0.49 carries no more evidential weight than independent-sample averaging",
      "boundToClaim": null,
      "boundToPaper": "Paper I"
    },
    {
      "falsifierId": "\u03c1",
      "proposition": "Correlated correction channels: the one-half exponent assumes corrections aggregate like independent samples \u00b7 Paper I and the synthesis",
      "criteria": "The ceiling is derived by taking one half as the best an internal corrector can reach, on the reasoning that accumulated corrections combine the way independent measurements do. That step fails in both directions. Positively correlated corrections repeat the same mistake and aggregate worse than independence, so the exponent falls below one half. Negatively correlated corrections aggregate better than independence, which is the textbook antithetic-variates result, so an exponent above one half is permitted and the ceiling moves. This condition fires, and the specific value of two is withdrawn, if a pre-specified measurement of the correction exponent on a system whose correction channels are characterised returns a value materially away from one half in either direction while the derivation still asserts one half",
      "status": "OPEN",
      "statusNote": ", and raised by this programme against itself as the single most likely point of failure. The author\u2019s dated priors of 9 August 2026 rate the claim that the value is exactly two as a genuine gamble, materially less than even, and name this channel as the reason. A parallel-channel measurement near zero exists in the programme\u2019s own data, but it is not offered as support here, for two stated reasons: it measures a different object, whether many independent generations aggregate under majority voting rather than whether successive corrections of one trajectory reduce error near-independently, and one of the six models measured contradicts the pattern at 0.31 with an r-squared of 0.93. So the mechanism is named, the derivation\u2019s dependence on it is disclosed, and the deciding measurement is not yet filed. What follows from it is architectural: a corrector built on the same substrate as the system it corrects cannot be anti-correlated with itself, sharing its architecture, training distribution and blind spots, so one half is a ceiling for that class rather than a typical value, while a corrector drawn from a different class may exceed it. That is a claim about real systems and not a theorem, and it is the thing being tested",
      "boundToClaim": null,
      "boundToPaper": "Paper I"
    },
    {
      "falsifierId": "024",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 The conjunction priority kill: the priority claim dies to any earlier document that already holds components one to four",
      "criteria": "One earlier document, of any earlier date and by any author, in which the first four components already sit together. That invitation does not expire, and every corpus searched is named in the statement paper.",
      "status": "OPEN",
      "statusNote": "the invitation stands; the corpora searched are named; statement paper section 5/7",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "025",
      "proposition": "Tier 1 \u00b7 Decisive trials \u00b7 The corrector-class ratio and the same-class cap (the eclipse instrument): correction across classes has to scale faster than correction within one",
      "criteria": "The registered primary is the difference in correction exponents between the two corrector classes, with the ratio secondary behind a denominator floor. The prediction is refuted by the reverse ordering, the ANTI-DIRECTIONAL verdict, or by a class difference whose interval lies wholly below the registered support threshold. An interval covering the null at the registered precision is INCONCLUSIVE, the roughly one-in-five outcome the registration prices in advance, and is reported as a failed attempt to confirm rather than as a refutation; the registration also discloses that the affirmative equivalence verdict does not fire under realistic between-target heterogeneity on this roster. Separately, one same-class corrector whose leverage measures wholly above one half refutes the same-class cap and moves the value of the ceiling; the ceiling relation itself stands",
      "status": "OPEN",
      "statusNote": "drafted registered instrument awaiting human submission; the author's own pilot leans towards the null at present and states as much",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "026",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 Law III sets out a relation between quantities measurable apart from each other, and is not a definition",
      "criteria": "Wherever both quantities can be obtained, the relation breaks if alpha (read off growth trajectories) and gamma (read off paired capability-correction protocols), each measured on its own, fail to satisfy alpha_crit = 1/(1-gamma) (the corrected 16 August 2026 form); missing is something a relation can do and a definition cannot",
      "status": "UNTESTED",
      "statusNote": "no measurement of gamma exists yet; what is claimed is the relation, while the value two remains the conjecture",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "027",
      "proposition": "Tier 1 \u00b7 Decisive trials \u00b7 Corrector-class audit, H1 (study-ac): how the determined-exponent case fares in systems already deployed",
      "criteria": "Falls if the audit turns up class 3 (determined-exponent) mechanisms in production deployments at more than 10 per cent of the inventory; that result would beat confirmation for interest, since beta_X would then be open to field measurement straight away",
      "status": "OPEN",
      "statusNote": "drafted registration awaiting human submission",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "028",
      "proposition": "Tier 1 \u00b7 Decisive trials \u00b7 Corrector-class taxonomy exhaustiveness, H3 (study-ac): no mechanism may fall outside the four classes",
      "criteria": "A single mechanism the decision tree cannot place is enough to refute it; a fifth dependency would collapse the elimination argument in Paper XIII section 8, and withdrawal of that argument is pre-committed",
      "status": "OPEN",
      "statusNote": "drafted registration awaiting human submission; the withdrawal consequence is registered",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "029",
      "proposition": "Tier 1 \u00b7 Decisive trials \u00b7 Genesis-governor joint necessity, H1 (study-u)",
      "criteria": "Survives only where the both-arm comes out above each of the remaining three arms AND the interaction contrast is positive with its 97.5 per cent interval clear of zero; let either condition go and joint necessity is refuted, and the author's own registration names the adverse result as the most likely one",
      "status": "OPEN",
      "statusNote": "drafted registration awaiting human submission",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "030",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 There has to be an R-star crossover (foundational F7)",
      "criteria": "Find no linear-to-super-linear transition anywhere and the transitional-regime prediction is falsified; R-star is the mechanistic marker that sets recursive amplification apart from plain redundancy",
      "status": "UNTESTED",
      "statusNote": "",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "031",
      "proposition": "Tier 1 \u00b7 Measured legs \u00b7 Sequential comes out above parallel (foundational F1)",
      "criteria": "Should alpha_seq sit at or beneath alpha_par consistently from one system to the next, the compositional-mechanism claim is refuted",
      "status": "OPEN",
      "statusNote": "so far it has held without exception across the six blinded models; any new system can refute it",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "032",
      "proposition": "Tier 1 \u00b7 Measured legs \u00b7 Per-domain sequential advantage, H1 (paper-ii registration)",
      "criteria": "Support requires the bootstrap 95 per cent interval on the sequential-minus-parallel exponent difference to sit wholly above zero in EVERY domain tested; one domain is enough to lose it, which is exactly why it was registered",
      "status": "OPEN",
      "statusNote": "drafted registration awaiting human submission",
      "boundToClaim": null,
      "boundToPaper": "Paper ii"
    },
    {
      "falsifierId": "033",
      "proposition": "Tier 1 \u00b7 Measured legs \u00b7 Embedded wins over post-hoc on both thresholds at once, H2 (study-v)",
      "criteria": "Miss either registered threshold and it is refuted: drift in the post-hoc arm has to come out above 15 per cent AND the deepest-layer arm has to remain under 5 per cent",
      "status": "OPEN",
      "statusNote": "drafted registration awaiting human submission",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "034",
      "proposition": "Tier 1 \u00b7 Measured legs \u00b7 Leaf venation at d = 2 (foundational F12)",
      "criteria": "The biological derivation is refuted wherever alpha in leaf-venation networks departs significantly from 1.5. BOUNDARY NOTE: it stands as an open verification item, recorded at build, whether 1.5 is derived in-house or is instead the West-family reciprocal form (d+1)/d; this row goes out with that question posed rather than settled",
      "status": "UNTESTED",
      "statusNote": "the provenance of the derivation is flagged for checking against foundational section 5.4",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "035",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 The identity takes a multiplicative form (make it additive and the ARC Equation dies)",
      "criteria": "Show intelligence and recursion to combine additively instead of multiplicatively and U = I x R^alpha is itself dead",
      "status": "OPEN",
      "statusNote": "",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "036",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 The mechanism: the exponent as measured has to agree with alpha = 1/(1-beta)",
      "criteria": "Where the measured exponent departs systematically from what the composition parameter predicts, the mechanism is refuted, and that holds even in cases where the form itself survives",
      "status": "OPEN",
      "statusNote": "",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "037",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 Nothing outside the ARC family fits better",
      "criteria": "The family claim goes should a functional form from outside the family fit the test-time compute data materially better across models",
      "status": "OPEN",
      "statusNote": "",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "038",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 Withdrawal on channel disagreement (study-ag): the gamma-structure unification",
      "criteria": "Disagreement between the registered channels pulls the unification off every surface it has reached, given the same prominence the framing received",
      "status": "OPEN",
      "statusNote": "drafted registration awaiting human submission; the withdrawal is pre-committed",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "039",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 Paper X F1: a phase boundary has to be there",
      "criteria": "Where long-run behaviour shifts smoothly as the compounding coupling changes and no threshold turns up, the stability law's claim about a boundary is dead",
      "status": "OPEN",
      "statusNote": "a sharp boundary is what the model predicts, and smooth variation kills it",
      "boundToClaim": null,
      "boundToPaper": "Paper X"
    },
    {
      "falsifierId": "040",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 Paper X F2: what decides is the margin and not raw speed",
      "criteria": "Should divergence follow raw speed instead of the scaling margin, the fast diverging and the slow converging whatever the coupling, the co-scaling reading is dead",
      "status": "OPEN",
      "statusNote": "",
      "boundToClaim": null,
      "boundToPaper": "Paper X"
    },
    {
      "falsifierId": "041",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 Paper X F6: the spectral threshold (Theorem 5)",
      "criteria": "Find the correction operator's null axis suppressed in E6 as well, and the spectral threshold theorem is false, with misalignment failing to behave the way the model says it does",
      "status": "OPEN",
      "statusNote": "",
      "boundToClaim": null,
      "boundToPaper": "Paper X"
    },
    {
      "falsifierId": "042",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 Reciprocity and the cap-boundary joint test (study-ae H1/H2)",
      "criteria": "H1 goes if the combined interval sits entirely outside plus-minus delta in any primary same-class cell; H2 goes jointly if either the cap interval or the boundary interval sits entirely above its registered value in any same-class cell",
      "status": "OPEN",
      "statusNote": "drafted registration awaiting human submission; the instruments have to clear the separability battery before anything else",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "043",
      "proposition": "Tier 1 \u00b7 The laws \u00b7 Gamma constancy versus the corridor derivative (study-af H1)",
      "criteria": "This is the registered instrument for sign(dgamma/dalpha): a slope interval falling clear of the ZERO_TREND_BAND of 0.05 in slope units establishes that constancy has been departed from and fixes the direction, while a slope interval inside the band reads as constancy on that substrate",
      "status": "OPEN",
      "statusNote": "drafted registration awaiting human submission; this single measurement fixes both the ceiling's direction and the stability class it belongs to",
      "boundToClaim": null,
      "boundToPaper": null
    },
    {
      "falsifierId": "044",
      "proposition": "Tier 2 \u00b7 Eden Protocol engineering \u00b7 The monitoring-removal test (F2): the delta between embedded and external",
      "criteria": "Prediction F2 is falsified where the measured embedded-versus-external delta comes out no different across the four registered models",
      "status": "OPEN",
      "statusNote": "an engineering test registered at a milestone",
      "boundToClaim": null,
      "boundToPaper": null
    }
  ]
}
