{ "schema_version": 3, "survey": { "title": "Impossibility Results in AI: A Survey", "authors": [ "Mario Brcic", "Roman V. Yampolskiy" ], "doi": "10.1145/3603371", "arxiv": "2109.00484v2", "inventory_basis": "Table 1 in arXiv v2", "expected_result_count": 44 }, "vocabulary": { "relationship": [ "EXACT", "EQUIVALENT", "RELATED", "DEPENDENCY_ONLY", "UNCLEAR" ], "lean_artifact_type": [ "REFERENCE", "WRAPPER", "NEW_PROOF", "BRIDGE" ], "spdx_license": [ "Apache-2.0", "BSD-3-Clause", "MIT" ], "ai_interpretation_status": [ "HUMAN_REVIEW", "STATEMENT_REVIEWED", "REVIEWED" ], "relation_kind": [ "BOUNDARY_PARTNER", "BUILDS_ON", "INSTANTIATES", "REFINES" ], "result_shape": [ "ACHIEVABILITY", "BOUND", "CHARACTERIZATION", "INFRASTRUCTURE", "POINT_IMPOSSIBILITY" ], "tag": [ "agent-incentives", "algorithmic-information", "compositionality", "computability", "computational-complexity", "control-theory", "decision-theory", "ethics", "information-theory", "interpretability", "learning-theory", "multi-agent", "oversight", "preference-inference", "provability-logic", "social-choice", "verification" ], "escape_axis": [ "RESTRICT_CLASS", "ADD_INFORMATION", "RELAX_EXACTNESS", "MOVE_TO_PRIOR", "ADD_RESOURCE", "WEAKEN_UNIFORMITY" ], "escape_route_status": [ "FORMALIZED", "STATED", "NAMED_ONLY" ], "statability": [ "EXTERNAL_ONLY", "TRIAGED_DISTINCT", "CANDIDATE_LEAD", "BLOCKED_ON_PRIMITIVE", "UNTRIAGED" ] }, "source_catalog": { "abramsky-zvesper-2010-lawvere-bk": { "citation": "S. Abramsky and J. Zvesper, “From Lawvere to Brandenburger-Keisler: interactive forms of diagonalization and self-reference,” arXiv:1006.0992, Jun. 2010.", "locator": "https://arxiv.org/abs/1006.0992", "role": "work", "retrieved": "2026-08-11", "notes": "Recasts Brandenburger–Keisler as a fixed-point argument in regular categories, reducing it to a relational form of Lawvere's one-agent diagonal argument. Catalogued because it is the stated bridge between CLM-LAWVERE-CCC-001 and brandenburger-keisler-2006, and because it separates the type-level Lawvere statement the atlas holds from the relational one the interactive cases need. It does not follow that regular-category machinery is required to formalize them: section 5.2 reformulates weak and very weak point surjectivity relationally, and the section 7 composition analysis is conducted concretely over relations on sets. No proof-assistant formalization found on 2026-08-11." }, "aisi-alignment-project-2026": { "citation": "The Alignment Project (UK AI Security Institute), research agenda — eleven priority research areas broken down by mathematical discipline, each listing open subproblems: benchmark design, cognitive science, computational complexity theory, economic and game theory, monitoring and red-teaming, evaluation and guarantees in RL, information theory and cryptography, interpretability, learning theory, post-training and elicitation, probabilistic methods.", "locator": "https://alignmentproject.aisi.gov.uk/research-agenda", "role": "directory", "retrieved": "2026-08-07", "notes": "Directory of research areas, not of individually numbered conjectures; each area states subproblems in prose. Reached via mathforaisafety-2026. The computational-complexity-theory area (scalable oversight, debate, prover-verifier games) is the one nearest current atlas surfaces." }, "atlas-ref-cohen-sharir-shashua-2016": { "citation": "N. Cohen, O. Sharir, and A. Shashua, “On the expressive power of deep learning: a tensor analysis,” Conference on Learning Theory (COLT), 2016.", "locator": "https://arxiv.org/abs/1509.05009", "role": "work", "notes": "Statement source for LAND-DL-001, reproduced through the Archive of Formal Proofs entry Deep_Learning." }, "atlas-ref-everitt-2016-self-modification": { "citation": "T. Everitt, D. Filan, M. Daswani, and M. Hutter, “Self-Modification of Policy and Utility Function in Rational Agents,” in Artificial General Intelligence (AGI 2016), Lecture Notes in Computer Science, vol. 9782, pp. 1-11, Springer, 2016. doi: 10.1007/978-3-319-41649-6_1. Technical report with the proofs: arXiv:1605.03142.", "locator": "https://doi.org/10.1007/978-3-319-41649-6_1", "role": "work", "notes": "Statement source for LAND-GOAL-001. VERSION: the published chapter states its theorems and proves none of them - Proofs for all theorems are provided in a technical report, namely arXiv:1605.03142 - so the published text is canonical for statements and the technical report is the only place any proof exists. The two number the same results differently. Published Theorem 12 (realistic policy-modifying agents make safe modifications) is the atlas target and is word-for-word the report Theorem 16; published Theorems 10 and 11 are report Theorems 14 and 15; published Definitions 1, 3, 7, 8, 9 are report Definitions 3, 5, 10, 11, 12. The report alone carries Lemma 13, Definition 18, Lemma 19, Theorem 20 and Theorem 21, which is proof apparatus the chapter does not print rather than material it dropped: the report proof of Theorem 16 invokes its Appendix-A Theorem 20 as its first step." }, "atlas-ref-everitt-hutter-2016-vrl": { "citation": "T. Everitt and M. Hutter, “Avoiding Wireheading with Value Reinforcement Learning,” arXiv:1605.03143v1 [cs.AI], 10 May 2016. Published in Artificial General Intelligence (AGI 2016), Lecture Notes in Computer Science, vol. 9782, pp. 12-22, Springer, 2016.", "locator": "https://arxiv.org/abs/1605.03143v1", "role": "work", "retrieved": "2026-09-09", "notes": "Statement source for LAND-VRL-001. VERSION: only arXiv:1605.03143v1 has been read; it is pinned at sha256 a0a9a8c96a52131f297ee123fcabd8a11a24641a640d35592cd688f662fd37a9 privately, manifested 2026-09-09. Every statement number the atlas uses for this work is the arXiv version's. Corrected 2026-09-10: this said the Springer chapter 'is closed and has NOT been obtained, so no claim is made that its numbering agrees', which was stale. docs/provenance/source-coverage-audit.md records that the published chapter WAS obtained the same day, at pp. 12-22 of LNAI 9782, sha256 4486e912ea2c0ed387bfaa4eaaf431db39ae2f362902fb880f5f2bc14d18d403, and that the two versions do not renumber: Definitions 2/3/5/7/8/9/10/11/12, Assumptions 4/6/15, Lemma 13 and Theorem 14 carry the same numbers in both, and equation (1) is equation (1) in both with identical text. The chapter carries no appendix, which is why the Lemma 27 row is graded against a statement the chapter delegates outward. Companion of arXiv:1605.03142 - consecutive identifiers, same day, same venue - and the two are cross-referenced in both directions." }, "atlas-ref-gjt-2013-truly-playable": { "citation": "V. Goranko, W. Jamroga and P. Turrini, “Strategic games and truly playable effectivity functions,” Autonomous Agents and Multi-Agent Systems 26(2): 288-314, 2013. doi: 10.1007/s10458-012-9192-y. Conference version: AAMAS 2011.", "locator": "https://doi.org/10.1007/s10458-012-9192-y", "role": "work", "retrieved": "2026-09-09", "notes": "Statement source for LAND-SOV-TRULYPLAYABLE-001. VERSION: the published text is subscription-only and has NOT been obtained; the author manuscript of the JAAMAS version is pinned at sha256 9fc20337290d725d50f60772642fb942f36321f382c979b9a57c6f5e492a965f privately, manifested 2026-09-09, and every statement number this atlas uses for this work is that manuscript's. The AAMAS 2011 conference version is pinned beside it at sha256 05d8bf71cc41b48ed4565bea8f87cae4051ee32004b3003a78f202539892f54d and is not the text graded. Venue, volume, pages and DOI are from the publisher record rather than from the document. TEXT EXTRACTION IS UNSAFE on this file: it drops the complement bar in the section 3.1 counterexample, turning the cofinite sets into the finite ones, so statements were read from rendered pages 5, 6, 13 and 14. This paper refutes the left-to-right direction of Pauly 2002 Theorem 3.2, which AISafetyAtlas.Sovereignty.PlayableConverse carries, and supplies the repair graded in section 14 of docs/provenance/source-coverage-audit.md." }, "atlas-ref-majumdar-2026": { "citation": "S. Majumdar, “The Relativity of AGI: Distributional Axioms, Fragility, and Undecidability,” arXiv:2601.17335, Jan. 2026.", "locator": "https://arxiv.org/abs/2601.17335", "role": "work", "retrieved": "2026-08-12", "notes": "Related-literature audit only; no claim row or paper-fidelity grade. The intended certification argument is Rice-shaped, but semanticity and nontriviality on an arbitrary effectively presented admissible-agent subclass do not suffice for Rice's theorem without an acceptable numbering, universality, or a property-preserving reduction. The self-certification corollary adds no independent self-reference mechanism, and the Tarski-style proof additionally assumes unprovided completeness for negative instances. See docs/guide/related-literature.md." }, "atlas-ref-manheim-garrabrant-2018-goodhart-variants": { "citation": "D. Manheim and S. Garrabrant, “Categorizing Variants of Goodhart's Law,” arXiv:1803.04585v4, 2019.", "locator": "https://arxiv.org/abs/1803.04585v4", "role": "work", "retrieved": "2026-09-11", "notes": "Model source for LAND-GOODHART-SELECTION-001. Pinned at sha256 7691af8a05ed88262d0eaebebf082f5d028979445a0cd586acc29c861a7a012a privately, manifested 2026-09-10. READ IN FULL, all ten pages, from rendered images of pages 1 to 9, and graded in section 17 of docs/provenance/source-coverage-audit.md. THIS PAPER NUMBERS NO THEOREM, PROPOSITION, LEMMA OR COROLLARY OF ANY KIND - verified by scanning the whole document. What it numbers is nine equations, each a generative 'Simple Model' with its consequences asserted in running prose. The two Lean modules are therefore routed as ATLAS-ORIGINAL work citing this paper for its model, per docs/provenance/by037-by038-goodhart-campbell-plan.md, and nothing books an unnumbered gloss as coverage of a printed theorem. TWO THINGS PRINT SAYS TWO WAYS. The page-2 setup writes the permissible region non-strictly, s in A if M(s) >= c, while section 1 writes 'the values of G when M > c'; the two modules each follow the passage they formalize. And equations (8) and (9) are printed IDENTICALLY, G_{A_R} = G_{A_0} + M_R, for two variants print distinguishes only in prose. Section 1 also says 'despite the lack of bias' beside a formula whose noise carries a mean mu; the atlas grades against the formula and leaves the mean free." }, "atlas-ref-melo-2024": { "citation": "G. A. Melo, M. R. O. A. Máximo, N. Y. Soma, and P. A. L. Castro, “On the Undecidability of Artificial Intelligence Alignment: Machines that Halt,” arXiv:2408.08995, 2024; also Sci. Rep., 2025.", "locator": "https://arxiv.org/abs/2408.08995", "role": "work", "notes": "STATEMENT SOURCE for AISafetyAtlas.Verification.AgentBehavior's theorem, and for LAND-VERIF-AGENTBEHAVIOR-001. Graded on 2026-09-11 in section 24 of docs/provenance/source-coverage-audit.md: 5 Yes, 0 Partial, 5 No, 1 Beyond. CORRECTED 2026-09-11: this note used to call the paper 'Related literature' and the module header named Rice 1953 as the source. Both roles are real and the header had them reversed. This paper STATES the claim the module's theorem renders; Rice's theorem is the PROOF ROUTE, reached through Mathlib's ComputablePred.rice rather than through this paper. It remains true that this is NOT a Brcic-Yampolskiy Table-1 source - that is a fact about the survey, not about which work states the claim. PINNED as melo-maximo-soma-castro-arxiv-2024-undecidability-of-ai-alignment-machines-that-halt.pdf, sha256 66eb3448f8f36602d71f65c155c4b6e32de3c2f92ad88590583bb6a225f8723c, manifested 2026-09-11, arXiv:2408.08995v1, 7 pp. Statements read from rendered pages 1, 3 and 4. Also published in Sci. Rep. (2025); that version is not pinned and no grade rests on it. IT NUMBERS NOTHING - no theorem, definition, lemma or proposition in seven pages - so it is graded the way sections 17 and 20 grade unnumbered printed claims. The claim is the abstract's first sentence: whether an arbitrary AI model satisfices a non-trivial alignment function of its outputs given its inputs is undecidable. FOUR THINGS ITS 'Formal Proof' SECTION SUPPLIES, all matching the module: the objects are PARTIAL functions (its reduction says so explicitly, which matters because the title is 'Machines that Halt'); extensionality is stated (two TMs accepting the same language both satisfice or neither); non-triviality comes WITH BOTH WITNESSES, a dummy-positive TM and a dummy-negative TM; and it says 'decides', i.e. total, sound and complete, which is what BehavioralSafetyVerifier's three fields are. TWO ROUTES, ONE TAKEN: print offers a restatement of Rice AND an explicit Halting-Problem reduction. The atlas takes the first; no halting bridge exists in the tree. NOT HERE: the enumerable set of architecturally aligned models built from 'axioms' (an architecture argument, not a theorem - print exhibits no enumeration and proves no closure), the special decidable case for finite-input D-ANNs, the masking architecture of Figure 2, and the proposed halting constraint on the judge function.", "retrieved": "2026-09-11" }, "atlas-ref-parent-benzmueller-condnorm": { "citation": "X. Parent and C. Benzmüller, “Conditional normative reasoning as a fragment of HOL,” formalized in the Archive of Formal Proofs entry CondNormReasHOL.", "locator": "https://isa-afp.org/entries/CondNormReasHOL.html", "role": "work", "notes": "Statement source for LAND-PE-001 (Åqvist system E with the Parfit mere-addition use case)." }, "atlas-ref-pauly-2002-coalition-logic": { "citation": "M. Pauly, “A Modal Logic for Coalitional Power in Games,” Journal of Logic and Computation 12(1): 149-166, 2002. doi: 10.1093/logcom/12.1.149.", "locator": "https://doi.org/10.1093/logcom/12.1.149", "role": "work", "retrieved": "2026-09-09", "notes": "Statement source for LAND-SOV-PLAYABILITY-001. Pinned at sha256 aa761c97e227ec186beb2e060973a829f51b9c3ef6bb3ed5184e7e36cd17826d privately, manifested 2026-09-09. TEXT EXTRACTION IS UNSAFE: it strips every formula from this file, leaving prose with holes, so statements were read from rendered pages 152 and 155. THE CONVERSE HALF OF THEOREM 3.2 IS FALSE at an infinite state set - refuted by Goranko, Jamroga and Turrini 2013, atlas-ref-gjt-2013-truly-playable, whose cofinite counterexample AISafetyAtlas.Sovereignty.PlayableConverse carries. Theorem 3.3 inherits that defect, because print's proof of its surviving direction invokes Theorem 3.2. Graded statement by statement in section 15 of docs/provenance/source-coverage-audit.md." }, "atlas-ref-ring-orseau-2011-delusion": { "citation": "M. Ring and L. Orseau, “Delusion, Survival, and Intelligent Agents,” in Artificial General Intelligence (AGI 2011), Lecture Notes in Artificial Intelligence, vol. 6830, pp. 11-20, Springer, 2011. doi: 10.1007/978-3-642-22887-2_2.", "locator": "https://hal.science/hal-01000226v1", "role": "work", "retrieved": "2026-09-09", "notes": "Statement source for LAND-WIRE-OBJ-001. VERSION: the Springer chapter is closed - OpenAlex reports oa_status closed and any_repository_has_fulltext false, Unpaywall reports is_oa false with no oa_locations. What is pinned is the AUTHOR DEPOSIT on HAL, hal-01000226v1, https://hal.science/hal-01000226/document, sha256 a207ab73c87896e0663e4998906aa4c4bcef6d93a43d8e29ebc3cdb81e32ea6d privately, manifested 2026-09-09; HAL is Orseau's institutional repository through UMR AgroParisTech 518 / INRA, so this is a lawful green route. It is the author version and has NOT been compared with the publisher's typeset chapter; pagination differs (HAL pp. 1-10, Springer pp. 11-20). The HAL version's equations (1) to (3) use the a_{t_h}, t_h = |h| + 1 indexing. The paper's seven numbered items are Statements 1 to 7, each supported by an Arguments paragraph rather than a proof; it prints no theorem, lemma or definition environment." }, "atlas-ref-shalev-shwartz-ben-david-2014": { "citation": "S. Shalev-Shwartz and S. Ben-David, Understanding Machine Learning: From Theory to Algorithms. Cambridge University Press, 2014.", "locator": "https://www.cs.huji.ac.il/~shais/UnderstandingMachineLearning/", "role": "work", "notes": "Statement source for the no-free-lunch theorem reproduced as LAND-NFL-001 (Theorem 5.1)." }, "atlas-ref-turner-tadepalli-2022-retargetable": { "citation": "A. M. Turner and P. Tadepalli, “Parametrically Retargetable Decision-Makers Tend To Seek Power,” arXiv:2206.13477v2, 2022. Published in Advances in Neural Information Processing Systems 35 (NeurIPS 2022).", "locator": "https://arxiv.org/abs/2206.13477v2", "role": "work", "retrieved": "2026-09-09", "notes": "Statement source for LAND-SOV-RETARGETABLE-001 and LAND-DEC-MDP-001. VERSION: the NeurIPS 2022 version, DOI 10.52202/068431-2276, is NOT open access and has not been obtained, so the pinned text is arXiv v2, sha256 5e4e811bd5a8cbf62203f29e9b7e430d0597d123397ab9f2be12ade6878a694e privately, manifested 2026-09-09, and EVERY statement number this atlas uses for this work is arXiv v2's. The file's xref table is damaged and needs reconstruction; formulas are AMS-glyph and text extraction is unsafe for them, so statements were read from rendered pages 4, 5, 12, 14-21 and 35. CITATION SHAPE: 'appendix B.2' means the SUBSECTION 'B.2 Helper results on retargetable functions', which holds lemmas B.7 to B.11 - not lemma B.2, a different statement in subsection B.1. ONE CORRECTION OF PRINT: lemma B.7's item 1 is printed with a guard that print's own proof cannot have, because equation (25) applies item 1 at the parameter phi_i-inverse dot theta-star, where the guard is the negation of what equation (29) concludes; lemma B.9, print's only consumer of B.7, discharges item 1 with no antecedent. The atlas states the unguarded form. Graded statement by statement in section 16 of docs/provenance/source-coverage-audit.md." }, "atlas-ref-wang-dorchen-jin-2026": { "citation": "C. L. Wang, K. Dorchen, and P. Jin, “On The Statistical Limits of Self-Improving Agents,” arXiv:2510.04399v2, Feb. 2026.", "locator": "https://arxiv.org/abs/2510.04399", "role": "work", "retrieved": "2026-08-12", "notes": "Related-literature audit only; no claim row or paper-fidelity grade. The paper proposes a VC-capacity boundary for self-modification, but the displayed iff under A1–A7 is not supported as stated: its sufficiency proof additionally assumes all reachable classes lie in one fixed finite-VC reference family, which does not follow from a uniform bound on the VC dimension of each class. The Two-Gate oracle inequality also compares ERM in the final subclass with the infimum over the larger reference family without an approximation or optimization premise. See docs/guide/related-literature.md." }, "atlas-ref-zhuang-hadfield-menell-2020-misaligned-ai": { "citation": "S. Zhuang and D. Hadfield-Menell, “Consequences of Misaligned AI,” Advances in Neural Information Processing Systems 33 (NeurIPS 2020).", "locator": "https://proceedings.neurips.cc/paper/2020/hash/b607ba543ad05417b8507ee86c54fcb7-Abstract.html", "role": "work", "retrieved": "2026-09-11", "notes": "Statement source for LAND-GOODHART-OVEROPT-001. TWO VERSIONS, BOTH PINNED, AND THE PUBLISHED ONE IS CANONICAL. The NeurIPS 2020 version is at sha256 bb5c7b5d179c6d63a777ef946c0449adeffc036036883280248ce0e50ae32710 and states every result, but says 'proofs for all theorems and propositions can be found in the supplementary material', which is not in that file. The arXiv preprint at sha256 37723b9165855628c3b0dd11908884dde8edd145ae277017be861d40dc8724aa carries the proofs inline, and Theorem 1's four-line proof is read from it. Both are privately, manifested 2026-09-10; statement numbering agrees between them for everything graded. Statements read from rendered pages 3 to 8 of the published version and page 5 of the preprint. Graded in section 18 of docs/provenance/source-coverage-audit.md. THEOREM 1 AS PRINTED IS FALSE, and the atlas states the repaired form: print writes 'based on J < L attributes', which bounds the count above and not below, and its setup lets the proxy attribute set be empty. At J = 0 the proxy is constant, 'strictly increasing' is vacuously true of it, every feasible state attains the supremum, a constant sequence converges, and the conclusion fails at any state above its floors. Print's own proof requires the set inhabited - it picks 'some feature j' in it to raise - and PRINT STATES THE MISSING HYPOTHESIS ITSELF at Proposition 2, 'for any non-empty set of proxy attributes'. A SECOND DEFECT IN THE SETUP: print says the feasible set is exactly the sublevel set of a constraint strictly increasing in each attribute AND that each attribute is bounded below; those are jointly unsatisfiable for a nonempty feasible set, since lowering a coordinate preserves the sublevel condition. The atlas carries the box as a second constraint, which is the reading that makes the setup consistent, and print's separate closedness assumption then becomes redundant rather than dropped." }, "brandenburger-keisler-2006": { "citation": "A. Brandenburger and H. J. Keisler, “An Impossibility Theorem on Beliefs in Games,” Studia Logica, vol. 84, no. 2, pp. 211–240, Nov. 2006, doi: 10.1007/s11225-006-9011-z.", "locator": "https://doi.org/10.1007/s11225-006-9011-z", "role": "work", "retrieved": "2026-08-11", "notes": "Interactive diagonalization: a belief model of a certain kind must have a hole, so some belief a player can hold about the game is absent from any model available to the players themselves. Catalogued as the interactive-self-reference member of the diagonal family alongside CLM-LAWVERE-001. No proof-assistant formalization found on 2026-08-11; the formalizations located are mathematical and category-theoretic rather than machine-checked. Evidence: docs/provenance/embedded-self-knowledge-landscape.md." }, "brcic-yampolskiy-2023": { "citation": "M. Brcic and R. V. Yampolskiy, “Impossibility Results in AI: A Survey,” ACM Computing Surveys, 2023, doi: 10.1145/3603371, arXiv:2109.00484.", "locator": "https://doi.org/10.1145/3603371", "role": "directory", "notes": "The survey itself; source for its three author-original results BY-042, BY-043, BY-044." }, "brcic-yampolskiy-2023-own-results": { "citation": "M. Brcic and R. V. Yampolskiy, “Impossibility Results in AI: A Survey,” ACM Computing Surveys, 2023, doi: 10.1145/3603371 — the authors' own results section (BY-042, BY-043, BY-044) with proof sketches.", "locator": "https://doi.org/10.1145/3603371", "role": "work", "notes": "Depth source for the survey's three author-original results. Distinct from the Table 1 directory entry brcic-yampolskiy-2023, which enumerates imported results and carries no statement to grade against." }, "breuer-1995-self-measurement": { "citation": "T. Breuer, “The Impossibility of Accurate State Self-Measurements,” Philosophy of Science, vol. 62, no. 2, pp. 197–214, Jun. 1995.", "locator": "https://philpapers.org/rec/BRETIO-3", "role": "work", "retrieved": "2026-08-11", "notes": "An observer cannot distinguish all present states of a system in which it is contained — classical or quantum, deterministic or stochastic time evolution. The primary PDF was read on 2026-08-11: §3.2 defines the inference map and exact measurement, §3.3 defines the restriction/proper-inclusion setup, §3.5 gives the meshing condition, and Propositions 1–2 state the abstract no-go results. Exactly one in-tree artifact is graded against this paper: LAND-SELFMEAS-002, the faithful abstract set-theoretic core, at RELATED rather than EXACT because physical apparatus construction, dynamics, the quantum treatment, and the EPR corollary are omitted. LAND-SELFMEAS-001 is a generic knowability specialization that is deliberately UNGRADED and carries no original_source_refs: it replaces the restriction map, inference map and meshing conditions with a product decomposition chosen to make a statement, so there is no second statement for a grade to be a grade of. PDF locator: https://cqi.inf.usi.ch/qic/Breuer95.pdf; catalogue locator remains the PhilPapers record. Search on 2026-08-11 across Mathlib, Isabelle/AFP, Rocq/Coq, HOL4, HOL Light, Agda and the general web found no machine-checked formalization; an unsuccessful search is not proof that none exists. Evidence: docs/provenance/self-measurement-kernel.md. PINNED 2026-09-11, and it had not been: the PDF is at sha256 5e558352d034e2a1e18f83167a85377e3e5f566635122f315c8b414bf2d452ac -- see the private manifest of 2026-09-11 for the authoritative hash -- and until that date no dated manifest carried it, although every other source in the coverage audit has one. THE FILENAME IS MISLEADING: it reads PhilSci1993 while the document's own first page reads Philosophy of Science, Vol. 62, No. 2 (Jun., 1995), pp. 197-214, JSTOR stable URL http://www.jstor.org/stable/188430. The citation above is the correct one; the file is not renamed because the hash identifies it. THE PAPER NUMBERS ONLY PROPOSITION 1 AND PROPOSITION 2 - the inference map, proper inclusion and the meshing condition are defined in running prose inside numbered sections, and the Lemma and the Corollary are unnumbered - so any citation of the form 'Breuer, Definition N' is wrong, and one in the module ledger was corrected on that date. Graded statement by statement in section 19 of docs/provenance/source-coverage-audit.md." }, "carter-shnayderman-2018-impossibility": { "citation": "I. Carter and R. Shnayderman, “The Impossibility of ‘Freedom as Independence’,” Political Studies Review, 2018, doi: 10.1177/1478929918771452.", "locator": "https://doi.org/10.1177/1478929918771452", "role": "work", "retrieved": "2026-09-09", "notes": "Statement source for LAND-SOV-INDEPENDENCE-001, together with list-valentini-2016-freedom-as-independence. Graded on 2026-09-11 in section 22 of docs/provenance/source-coverage-audit.md. PINNED as carter-shnayderman-2018-earlyview-political-studies-review-impossibility-of-freedom-as-independence.pdf, sha256 bb5bf47c635036527739b5d65597f0a7b0e6386d5107ebd77fccff942fa07dff, manifested 2026-09-09, 11 pp. VERSION: this is the publisher's OnlineFirst text and NOT a final paginated version. Its own page 1 gives the pagination as '1-11' and its running head reads 'Political Studies Review 00(0)', so it carries no volume, issue or journal page numbers. Cite it by DOI, never by page. The article's own pages 4 and 5 carry the two theses and were rendered and read. NUMBERS NOTHING: both central claims are displayed italicised theses - 'The impossibility of republican freedom' on page 4 and 'The impossibility of freedom as independence' on page 5. Both are rendered as CONDITIONALS, because each antecedent is a claim about politics that print supplies and the atlas does not assert. THE ARGUMENT IS IN THE HYPOTHESES: the republican thesis needs a relevant world carrying an IMPERMISSIBLE constraint, because on Pettit's conception a power to interfere must be arbitrary and hence held with impunity, so everyday threats inside a republican state are punished and do not count. The independence thesis needs only a constraint, because List and Valentini drop arbitrariness. That difference in the two Lean hypotheses is exactly print's argument. NOT HERE: the Kramer probability-versus-sheer-possibility diagnosis. Sovereignty.Boundary answers it in prose for its own predicate - an all-profiles invariant is the right shape against an adversary that searches rather than samples - and that reply is a docstring, not a theorem." }, "chandy-lamport-1985": { "citation": "K. M. Chandy and L. Lamport, “Distributed Snapshots: Determining Global States of Distributed Systems,” ACM Transactions on Computer Systems, vol. 3, no. 1, pp. 63–75, Feb. 1985, doi: 10.1145/214451.214456.", "locator": "https://doi.org/10.1145/214451.214456", "role": "work", "retrieved": "2026-08-11", "notes": "A possibility result, catalogued as the escape boundary dual to the self-knowledge impossibility statements: a process can determine a consistent global state of a distributed system while the computation continues. It bounds how any embedded self-knowledge claim may be worded — the obstruction concerns exact knowledge of the present state, not knowledge of a consistent past cut. Machine-checked in Isabelle/HOL as AFP entry Chandy_Lamport (Fiedler and Traytel, July 2020, BSD License), whose session parent Ordered_Resolution_Prover pulls Coinductive and Nested_Multisets_Ordinals, so reproduction uses the full pinned AFP release rather than a single-entry archive." }, "chen-ju-agotnes-2026-actual-alpha-powers": { "citation": "Z. Chen, F. Ju, and T. Ågotnes, “Representation theorems for actual and alpha powers over general concurrent game frames without assuming independence of agents,” arXiv:2607.10567v1, 2026.", "locator": "https://arxiv.org/abs/2607.10567v1", "role": "work", "retrieved": "2026-09-09", "notes": "Statement source for the ActualPower layer of AISafetyAtlas.Sovereignty.Separations, hosted by LAND-SOV-POWER-001. Graded on 2026-09-11 in section 23 of docs/provenance/source-coverage-audit.md: 9 Yes, 0 Partial, 5 No, 1 Beyond. Fact 1 item 1 closed the same day: outcomesOf is the outcome set of a joint action and outcomesOf_anti_coalition is print's shrinking inclusion, with vetoGame_outcomesOf_strict showing it can be proper. PINNED as chen-ju-agotnes-arxiv-v1-2026-representation-theorems-actual-alpha-powers-general-cgf.pdf, sha256 c7aece726de1b7a7a106c6461ee968dbfe020f17851e32cde1fa5a0894f4e971, manifested 2026-09-09, 50 pp. Statements read from rendered pages 8 and 10. TWO FILES, TWO PAPERS: the literature directory also holds chen-ju-agotnes-arxiv-v2-...-two-agent.pdf, sha256 4cb9ce1ce71e5d60726376827853a9f1012c5326c5c2df4943b8a31008cf26ac. The v1/v2 in those filenames are NOT version numbers. That file is arXiv:2603.04160v2, 'Representation theorems for actual and alpha powers over TWO-AGENT general concurrent game frames' - a different submission with a different title. The atlas cites 2607.10567 and section 23 grades that file only. WHAT IS TAKEN: Definition 5's two effectivity functions (alpha is Forces and effectivity, actual is ActualPower - a set is an actual power exactly when it EQUALS the outcome set of some available action, which is ActualPower's two clauses) and Definition 9's three frame conditions. GameForm is print's SID case: independence holds because dependent functions on disjoint index sets combine, determinism because outcome is a function, seriality when the strategy types are inhabited. THE PHRASE 'AT A SINGLE STATE' WAS DROPPED ON 2026-09-20 AND IT WAS FALSE: with a one-point state set every alpha power is trivial. gameFrame in AISafetyAtlas.Sovereignty.ConcurrentGameFrame is the correct embedding and its state set is the OUTCOME TYPE with state-independent dynamics. THE FOUR NARROWER ROWS CLOSED THE SAME DAY, together with Definition 1, Definition 8, Fact 1 item 2, the eight classes and print's caution about them; the section is 14 Yes, 0 Partial, 2 No, 2 Beyond. Print confirms SID is the standard class and the one Pauly's representation theorem covers. A CITATION DEFECT THIS GRADING FOUND, AND IT WAS OURS: the ActualPower docstring cited this paper and then quoted, as the literature's objection to alpha-monotonicity, a sentence that is NOT in this paper. It is in the two-agent companion, where it is prefaced 'as pointed out by [BBE19]' - a third source, unpinned. The docstring is corrected. Separately, the forces_superadditive docstring said superadditivity is 'not available' in classes not imposing independence; print says the labels 'do not record which conditions are required to fail' and the classes 'are not intended to be disjoint', so the right word is 'not guaranteed'. Also corrected. NOT HERE: 24 further definitions, Theorems 1-8, Lemmas 1-14, Propositions 1-2, Corollary 1, Facts 1-2 and the examples - i.e. the representation theorems, which are what the paper is for. The atlas claims nothing about representation." }, "cover-thomas-2006": { "citation": "T. M. Cover and J. A. Thomas, Elements of Information Theory, 2nd ed. Hoboken, NJ: Wiley-Interscience, 2006.", "locator": "https://doi.org/10.1002/047174882X", "role": "work" }, "critch-tsimerman-2025": { "citation": "A. Critch and J. Tsimerman, “A Taxonomy of Omnicidal Futures Involving Artificial Intelligence,” arXiv:2507.09369, 2025.", "locator": "https://arxiv.org/abs/2507.09369", "role": "work", "notes": "Five exhaustive categories, made mutually exclusive by earliest-match; the paper states that preventing all five is logically necessary for human survival without writing the argument down. Not yet referenced by any result." }, "droste-jansen-wegener-2002": { "citation": "S. Droste, T. Jansen, and I. Wegener, “Optimization with randomized search heuristics—the (A)NFL theorem, realistic scenarios, and difficult functions,” Theoretical Computer Science, vol. 287, no. 1, pp. 131–144, 2002, doi: 10.1016/S0304-3975(02)00094-4.", "locator": "https://doi.org/10.1016/S0304-3975(02)00094-4", "role": "work" }, "everitt-etal-2021-agent-incentives": { "citation": "T. Everitt, R. Carey, E. D. Langlois, P. A. Ortega, and S. Legg, “Agent Incentives: A Causal Perspective,” Proceedings of the AAAI Conference on Artificial Intelligence, vol. 35, no. 13, pp. 11487–11495, 2021, doi: 10.1609/aaai.v35i13.17368.", "locator": "https://doi.org/10.1609/aaai.v35i13.17368", "role": "work", "retrieved": "2026-08-19", "notes": "Peer-reviewed CID source RE24 cites: single-decision single-utility causal influence diagrams, policy as the decision mechanism, expected utility. Ingredient for Skeleton, Policy, and Model.value. Not a source of local interventions or of the margin class." }, "gokhale-leanforcontrol-2026": { "citation": "A. Gokhale and contributors, LeanForControl — a Lean 4 + Mathlib formalization of linear-systems and control theory, commit c5cedca, 2026-09-01. Apache-2.0.", "locator": "https://github.com/AnandGokhale/LeanForControl", "role": "work", "retrieved": "2026-09-11", "notes": "FORMALIZATION SOURCE, NOT STATEMENT SOURCE. Four of the seven modules of AISafetyAtlas.LinearSystems - MatrixLemmas, Controllability, Observability and Hautus - are adapted from this repository's LinearSystems/ track; BlockBound, Dynamics and Flow are atlas-original and take nothing from it; each atlas file header carries the upstream file, its sha256, and the atlas changes, and AISafetyAtlas/Upstream/LICENSE-NOTICE carries the repository-level notice. Only that track was taken: the Stability, Comparison, ODEs and Dini trees rest on seven custom axioms, including an assumed Picard-Lindelöf existence-uniqueness, and would not pass this repository's axiom audit. Upstream toolchain leanprover/lean4 v4.30.0-rc2." }, "hautus-1969-controllability-observability": { "citation": "M. L. J. Hautus, “Controllability and observability conditions of linear autonomous systems,” Indagationes Mathematicae (Proceedings), vol. 72, no. 5, pp. 443–448, 1969. ISSN 1385-7258.", "locator": "https://research.tue.nl/en/publications/controllability-and-observability-conditions-of-linear-autonomous", "role": "work", "retrieved": "2026-09-11", "notes": "Printed source of the eigenvalue tests named isObservable_iff_hautus and isControllable_iff_hautus in LAND-LINSYS-001. NAMED, NOT PINNED: only the bibliographic record was obtained; the TU/e portal offers no full text and the ScienceDirect identifier tried returned 404. No coverage-audit section grades this source, and none is claimed. Attempt manifest: the private manifest of 2026-09-11 in the literature directory." }, "howard-matheson-2005-influence-diagrams": { "citation": "R. A. Howard and J. E. Matheson, “Influence Diagrams,” Decision Analysis, vol. 2, no. 3, pp. 127–143, 2005, doi: 10.1287/deca.1050.0020. Reprint of the 1984 paper.", "locator": "https://doi.org/10.1287/deca.1050.0020", "role": "work", "retrieved": "2026-08-19", "notes": "Classical influence-diagram source cited by RE24 for the CID formalism: chance, decision, and value nodes. Ingredient for Skeleton, Policy, and value, via the Everitt et al. 2021 / RE24 CID specialization." }, "igel-toussaint-2003": { "citation": "C. Igel and M. Toussaint, “Recent Results on No-Free-Lunch Theorems for Optimization,” arXiv:cs/0303032 [cs.NE], 2003.", "locator": "https://arxiv.org/abs/cs/0303032", "role": "work" }, "igel-toussaint-2004": { "citation": "C. Igel and M. Toussaint, “A No-Free-Lunch Theorem for Non-Uniform Distributions of Target Functions,” J. Mathematical Modelling and Algorithms, vol. 3, no. 4, pp. 313–322, 2004, doi: 10.1023/B:JMMA.0000049381.24625.f7.", "locator": "https://doi.org/10.1023/B:JMMA.0000049381.24625.f7", "role": "work" }, "iliad-ecosystem-2026": { "citation": "Iliad ecosystem, incubated organizations — a directory of groups working on interpretability, theoretical models, agency, and alignment.", "locator": "https://www.iliad.ac/incubated-organizations", "role": "directory", "retrieved": "2026-08-07", "notes": "Reached via mathforaisafety-2026." }, "kalman-1963-linear-dynamical-systems": { "citation": "R. E. Kalman, “Mathematical description of linear dynamical systems,” Journal of the Society for Industrial and Applied Mathematics, Series A: Control, vol. 1, no. 2, pp. 152–192, 1963.", "locator": "https://doi.org/10.1137/0301010", "role": "work", "retrieved": "2026-09-11", "notes": "Printed source of the rank criteria named isObservable_iff_observabilityMatrix_rank_eq and its controllability dual in LAND-LINSYS-001. NAMED, NOT PINNED: not fetched, and the atlas grades nothing against it. Attempt manifest: the private manifest of 2026-09-11 in the literature directory." }, "lawvere-1969-diagonal-ccc": { "citation": "F. W. Lawvere, “Diagonal Arguments and Cartesian Closed Categories,” in Category Theory, Homology Theory and Their Applications II, Lecture Notes in Mathematics, vol. 92, pp. 134–145, Springer, 1969. doi: 10.1007/BFb0080769.", "locator": "https://doi.org/10.1007/BFb0080769", "role": "work", "notes": "Original source for the categorical Lawvere fixed-point theorem: a weakly point-surjective morphism into an exponential object forces the fixed-point property. The full CCC theorem is recorded as CLM-LAWVERE-CCC-001; no Atlas CCC formalization is claimed yet." }, "list-valentini-2016-freedom-as-independence": { "citation": "C. List and L. Valentini, “Freedom as Independence,” Ethics, vol. 126, no. 4, pp. 1043–1074, Jul. 2016, doi: 10.1086/686006.", "locator": "https://doi.org/10.1086/686006", "role": "work", "retrieved": "2026-09-10", "notes": "Statement source for LAND-SOV-INDEPENDENCE-001, together with carter-shnayderman-2018-impossibility. Graded on 2026-09-11 in section 22 of docs/provenance/source-coverage-audit.md, jointly with that paper: 11 Yes, 0 Partial, 7 No, 2 Beyond. PINNED as list-valentini-2016-published-ethics-freedom-as-independence.pdf, sha256 93fb7dad01fa93d20ce33550421943100cacc8ea3b9a9d541dce316bb2cc63d2, manifested 2026-09-09, published Ethics version, 32 pp. Statements read from rendered pages 4 and 5 = journal pages 1046 and 1047. NUMBERS NOTHING: the paper's content is a definition scheme, two named questions and a two-by-two matrix given as four numbered cases. Section II holds the scheme, the moralization question and the robustness question; section III holds the four cases and the attributions - case 1 Berlin, case 2 Nozick and Dworkin, case 4 Pettit's republican freedom, case 3 the paper's own freedom as independence. Footnote 8 forbids moralizing the notion of RELEVANCE, which is why constrains and permitted are separate fields. INDEPENDENCE OF THE TWO DIMENSIONS, CLOSED 2026-09-11: print asserts the two dimensions are independent, hence a genuine two-by-two. Examples.Sovereignty.four_corners_distinct is a Setting whose four conceptions are four different sets of free actions. NARROWER, AND PRINT SAYS SO: print states the two dimensions can be refined and made nonbinary, and that each case is a family of conceptions; Free takes two Bool parameters and forecloses both. NOT HERE: the functional-role desideratum, the positive case for case 3, and the political implications." }, "mais-a2-2026": { "citation": "Math for AI Safety (MAIS), research agenda MAIS-A2, “Behavioral tomography of causal world models” — the agenda in which MAIS-O23, O24, O25, O26, O27, O2, O28, O29, O30, O31, O32, O33, O34 and O35 are stated, together with the definitions they quantify over.", "locator": "https://github.com/lionellevine/MAIS/blob/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/agendas/A2/MAIS-A2.tex", "role": "work", "retrieved": "2026-08-20", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "content_sha256": "d61be3eed51f618dd3b9389693b14e066e89a9cef5e89985b4226fff658c3c4f", "notes": "The statement source every MAIS conjecture in conjectures.yaml is graded against except CONJ-025, whose printed problem is stated in agenda A3 (mais-a3-2026); the open-problems/*.md pages are one-page restatements pointing back into whichever agenda states them. Carries the definitions the atlas transcribes: def:cbn, def:local (local interventions, masks, profiles, mixtures), def:cid (binary CID — skeleton, model, policy pi: dom(O) -> [0,1], value, regret; the unmediated-task assumption is built into the definition rather than assumed on top of it), def:margin ((M1)-(M6)), def:twovar (the two-variable family MM_2(lambda)), prop:equiv. Chance variables and the decision are binary by definition, so the atlas's binary types are print's own restriction and not a narrowing. The rendered PDF at the same commit is sha256 91fcefc283db02f377c723120357bb3dc67836604986cfc17f6c0573bdaf6bed. Hashes and the per-statement label map are recorded in docs/provenance/mais-source-pin.md. The agenda is AI-written and not human-peer-reviewed." }, "mais-a3-2026": { "citation": "Math for AI Safety (MAIS), research agenda MAIS-A3, “The geometry and identifiability of superposition” — the agenda in which MAIS-O3, O36, O37, O38, O39, O40, O41, O42 and O43 are stated; MAIS-O38 is its Problem 4.8.", "locator": "https://github.com/lionellevine/MAIS/blob/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/agendas/A3/MAIS-A3.tex", "role": "work", "retrieved": "2026-08-27", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "content_sha256": "146f0cc95a0a5eb0cf3b2660c32d591169b0e346571b80ee63723a9906371387", "notes": "The statement source for the MAIS-O38 conjecture row, on the same footing MAIS-A2 has for the causal rows: the open-problems/*.md page is a one-page restatement pointing back into it, and at this commit the two agree word for word on Problem 4.8. Problem 4.8 carries the label prob:samples and is the only part of A3 the atlas transcribes; the surrounding superposition material — the estimator F_lambda, the recovery and merge notions, the smeared-feature and pentagon problems — is not. The problem's own vocabulary is self-contained: a matrix, k-sparse codes, the spark condition, permutation and invertible diagonal matrices, and Lebesgue measure on R^{n x m}, none of which is an A3 composite. The agenda is AI-written and not human-peer-reviewed." }, "mais-a6-2026": { "citation": "Math for AI Safety (MAIS), research agenda MAIS-A6 — the agenda in which MAIS-O70 is stated as prob:calibration, together with def:local, thm:aw and rem:conventions, the definitions it quantifies over.", "locator": "https://github.com/lionellevine/MAIS/blob/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/agendas/A6/MAIS-A6.tex", "role": "work", "retrieved": "2026-09-03", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "content_sha256": "3da2eda1b1fa9633d09e48c4ce3bab34bc22aeea4bfc72eeb04b6b919b1c1d3e", "notes": "Context source for CONJ-026, which is graded against mais-o70-2026 instead: the predicates that row grades are volume-order predicates and this artifact defines the invariant otherwise. def:local defines the local pair by zeta poles, then adds that it does not depend on the radius nor on inserting a smooth positive density, and that lambda is equivalently the exponent in the ball-volume asymptotic eq:volume; the atlas takes that equivalent volume form as primary and records the substitution rather than proving the equivalence. thm:aw states the four-case global pair for reduced-rank regression and glosses it, in the same sentence, as the minimum of the local coefficient over the zero fiber with multiplicity the largest among the minimizers; that gloss is why O70's third clause needs no separate Aoyagi-Watanabe hypothesis. rem:conventions fixes the compact domain and prior and excludes the vacuous case K identically zero, which for reduced-rank regression is exactly a vanishing ambient dimension. Its support clause guarantees only a neighborhood of *a* minimizing factorization, which is weaker than thm:aw's largest-multiplicity claim needs; the scalar model M=N=H=1, r=0 is the smallest witness. The agenda is AI-written and not human-peer-reviewed." }, "mais-a7-2026": { "citation": "Math for AI Safety (MAIS), research agenda MAIS-A7 — the singular-learning agenda containing MAIS-O7 and MAIS-O77 as Problems 3.7 and 3.9.", "locator": "https://github.com/lionellevine/MAIS/blob/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/agendas/A7/MAIS-A7.tex", "role": "work", "retrieved": "2026-09-05", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "content_sha256": "c645a51bf02b16077d14adce56e1c55a2fcb66f5004bbd0aec57fb01a2d54b52", "notes": "Context source for CONJ-027 and CONJ-028. CONJ-028 is graded against mais-o77-2026 instead, because the predicates it grades are volume-order predicates and this artifact defines the invariant otherwise. Definition def:llc defines the local invariant by the meromorphic continuation and leading pole of a zeta integral, then glosses the two-sided band volume, introduced with the words \"In words:\" and outside the definition environment. The atlas formalizes the operational two-sided volume order and does not prove the zeta-pole equivalence; that substitution is carried as the frontier A7-ZETA-BRIDGE (AISafetyAtlas.Conjectures.MAIS.A7ZetaVolumeBridge), with the results it buys stated conditionally in AISafetyAtlas/Conjectures/MAIS/A7Zeta.lean. No exact asymptotic constant is claimed either. Problem 3.7 conjectures monotonicity of the infimal coefficients along the critical-value ladder. Problem 3.9 asks for the local pair on the exact factorization fiber, whether it depends only on the two factor ranks, and the pair at every nonterminal critical set C_k. The target has distinct positive nonzero singular values and hidden width H greater than its rank. The agenda is AI-written and not human-peer-reviewed." }, "mais-issue-12-2026": { "citation": "Math for AI Safety (MAIS), issue #12, candidate complete solution of MAIS-O77, filed 2026-08-03.", "locator": "https://github.com/lionellevine/MAIS/issues/12", "role": "work", "retrieved": "2026-09-05", "content_sha256": "c0a2c0a8cb66858ca0160765b1465fb1f55bdd9be47edf79283d43e82af6c984", "notes": "Context for CONJ-028. The issue body points to the seven-page candidate PDF MAIS-O77-candidate-solution.pdf, sha256 5a4a919ffb30a110cd377e4df5c0a829c36ee11696bb8dc4211c6afde149a998, archived alongside the issue #3 attachment in the shared literature tree after a 2026-09-06 audit found the graded artifact retained nowhere locally and reachable only through a GitHub attachment URL. Part (a) explicitly overlaps the solution in MAIS issue #3 and relabels its multiplication-slice table; the atlas reuses the O70 implementation rather than duplicating it. The order-level table is machine-checked conditional only on EigenvalueLawStatement, the same Wishart frontier inherited from O70; the arithmetic minimum, the table-level minimizing-stratum predicate, and the extra maximum-tie-multiplicity assertion are unconditional. The minimal stratum is also checked at the germs, by isO77MinimizerCharacterization_o77Minimizers, which is conditional on EigenvalueLawStatement and on no further frontier. Part (b)'s all-saddle claim is now machine-checked unconditionally, as o77AllSaddlesHavePairOne_holds; part (a) remains conditional on EigenvalueLawStatement, so this source record represents only (b) as unconditionally verified. The issue claims no human referee review. Issue bodies are editable in place, so an edit invalidates the body hash.", "mais_solution": { "problem": "MAIS-O77", "verdict": "PARTIAL", "candidate_lean": "AISafetyAtlas.Conjectures.MAIS.O77AllSaddlesHavePairOne", "submitted": "A complete solution to both clauses: the local pair at every exact factorization together with the minimal stratum, and the pair at every point of every nonterminal critical set.", "checked": "Part (b) is proved unconditionally at print's own quantifiers, as o77AllSaddlesHavePairOne_holds, following the candidate's five printed steps rather than substituting a shortcut; two results it needs — the splitting of its equation (9) and its Lemma 2 — are absent from the Mathlib revision pinned here and were built here; NC-012 is that search, and it covers Mathlib alone, so it is not a claim about every formalization corpus. Part (a) is verified only under the eigenvalue-law frontier inherited from MAIS issue #3; that covers its minimal stratum too, checked at the germs rather than against the table. Four defects surfaced while checking, all of them the atlas's own -- each a statement that compiled, was axiom-clean, and was about something weaker than print. None was a defect in the submission; no error was found in it. The complete-solution claim is therefore not verified. The note's argument is followed, not replaced. Its equation (9) already writes the generalized splitting -- L - L(w) = Q_alpha(xi) + g(zeta) with g(0) = 0 and no nondegeneracy asked of g -- which is a Gromoll-Meyer splitting, and that is what was built. Only the name it gives the tool, the analytic Morse splitting lemma, is loose. Two departures are worth stating: the atlas proves the splitting C-infinity rather than analytic, which is all the note's Lemma 2 consumes, and loss_quartic_on_degenerateNull rules out a Morse-Bott normal form, which (9) does not claim and does not need. A 2026-09-06 audit closed the remaining prose step in part (a) by proving that the discrete minimisation the atlas computes with is the note's printed four-case closed formula, and pinned the note's own ten check-table values." } }, "mais-issue-3-2026": { "citation": "Math for AI Safety (MAIS), issue #3, candidate solution to MAIS-O70: local learning coefficients of reduced-rank regression, by Sneiderman, 2026.", "locator": "https://github.com/lionellevine/MAIS/issues/3", "role": "work", "retrieved": "2026-09-02", "content_sha256": "405c8cb324884607f3e827cdd915d8ac10ce01b969a7739616474b3fb8401cbe", "notes": "Context for CONJ-026. The hash is of the PDF attached to the issue, not of the issue page: the attachment carries the mathematical content and is the artifact graded. Theorem 1.1 gives the local pair at a rank stratum and Corollary 1.3 restates the residual threshold as a discrete minimisation over an integer index; both are transcribed as o70Pair, and the atlas closes that minimisation in residualMinCost_eq_argmin. Definition 2.1 fixes the local pair by exact ball-volume asymptotics “for some (equivalently, all but countably many) small delta”, and section 13 states that exactness at every radius would need subanalytic integration theory and is not claimed; the atlas's HasExactLocalPair matches that co-countable form. Section 13 also reports the support gap in the Aoyagi-Watanabe hypotheses that rem:conventions shares. The numerical receipts the issue cites are not reproducible from it. Not peer-reviewed.", "mais_solution": { "problem": "MAIS-O70", "filed_by": "Sneiderman", "verdict": "CONDITIONAL", "candidate_lean": "AISafetyAtlas.Conjectures.MAIS.o70Pair", "submitted": "A complete solution: the local learning coefficients of reduced-rank regression at every rank stratum, with the residual threshold restated as a discrete minimisation over an integer index.", "checked": "The rank table is transcribed as o70Pair and machine-checked, but under two propositions the candidate cites rather than derives: the real-Wishart eigenvalue law and the existence of exact local pairs. Both are recorded as frontiers with their own evidence, and NC-011 shows that none of the six baseline formalization corpora supplies the first. The discrete minimisation behind Corollary 1.3 is closed unconditionally, as residualMinCost_eq_argmin. The candidate's own section 13 states that exactness at every radius is not claimed, and the atlas matches that co-countable form rather than strengthening it." } }, "mais-issue-30-2026": { "citation": "Math for AI Safety (MAIS), issue #30, “MAIS-O38 candidate complete solution: m^3+2m fixed codes suffice for every sparsity k < m”, filed 2026-08-26 by 26david26.", "locator": "https://github.com/lionellevine/MAIS/issues/30", "role": "work", "retrieved": "2026-08-27", "content_sha256": "6e2db10eb10242c075ca331fcf87a604511b9b31df3d55a5c4b0d2d2d95d05ab", "notes": "Context for CONJ-025. Re-read 2026-08-30: the issue contains TWO theorems. Its main claim -- N = m^3 + 2m codes depending only on m and k suffice for every n >= 2k and almost every spark-condition A, at every 1 <= k < m -- is transcribed as AISafetyAtlas.Conjectures.MAIS.o38PolynomialSampleCandidate and proved as AISafetyAtlas.Examples.Conjectures.MAIS.o38PolynomialSampleCandidate_holds. It resolves CONJ-025 at every non-degenerate m; its codes being independent of n is the one remaining axis on which it is stronger than the graded claim. Its second theorem, labelled boundary, states that at m >= 2, k >= m and n >= 2k no finite list of k-sparse codes is uniquely coded at any spark-condition A, and observes that this settles the literal reading in which k < m is nowhere imposed. AISafetyAtlas.Examples.Conjectures.MAIS.not_uniquelyCoded_of_full_sparsity_spark proves the same mathematical boundary independently and without the n >= 2k hypothesis; the issue's statement and observation are credited. The issue says its mathematics was produced and checked entirely by AI systems with no human verification. Body sha256 as read 2026-08-27 and re-verified unedited 2026-08-30. The atlas machine-check comment is issuecomment-5466534850, updated 2026-08-30T04:08:15Z, body sha256 db8715b6dd9554bb25720766957901bc83ad9c0a8a37d6812965b722a9b2e138. Issue bodies and comments are editable in place, so later edits invalidate these records.", "mais_solution": { "problem": "MAIS-O38", "filed_by": "26david26", "verdict": "CHECKED", "candidate_lean": "AISafetyAtlas.Conjectures.MAIS.o38PolynomialSampleCandidate", "submitted": "A complete solution in two theorems: that a polynomial number of fixed codes depending only on the dimension and the sparsity suffice below the sparsity bound, and a boundary theorem above it.", "checked": "The main claim is transcribed and proved, and resolves the graded row at every non-degenerate dimension; its codes being independent of the ambient dimension is one axis on which it is stronger than the claim being graded. The boundary theorem is proved here independently and without one of its hypotheses. The atlas has posted this machine-check on the issue itself. The issue reports that its mathematics was produced and checked entirely by AI systems with no human verification." } }, "mais-issue-4-2026": { "citation": "Math for AI Safety (MAIS), issue #4, “MAIS-O34: exact fibers, regret geometry, and a single graph-threshold program”, filed 2026-08-03 by Robby955.", "locator": "https://github.com/lionellevine/MAIS/issues/4", "role": "work", "retrieved": "2026-08-20", "content_sha256": "f425da83395b457feb5615c9beed703675a977967890ebe1b97dd61efdd0b328", "notes": "The candidate complete criterion for MAIS-O34 part (a), transcribed by a conjecture row wherever that row is recorded, and the origin of the two-variable construction MAIS issue #6 credits. Part (b) — the first-order radius constant, the singular classification, and the graph-threshold program — is not transcribed. The issue's only author comment adds no mathematical claim: it says that LaTeX may not render on mobile and directs the reader to the PDF. Its body is preserved separately in docs/provenance/mais-source-pin.md for completeness, but it is not part of the graded candidate. Issue bodies are editable in place and carry no addressable revision, so the recorded hash is the body as read on 2026-08-20; an edit invalidates the grading resting on it. Open and unreviewed at that reading.", "mais_solution": { "problem": "MAIS-O34(a)", "filed_by": "Robby955", "verdict": "CHECKED", "candidate_lean": "AISafetyAtlas.Conjectures.MAIS.maisO34_exactFiberCandidate", "submitted": "A complete criterion for when the global behavioural fibre is a singleton, on the source's real three-parameter two-variable chart, together with a first-order radius constant, a singular classification, and a graph-threshold program.", "checked": "The criterion is transcribed at the source's own quantifiers and proved, as maisO34_exactFiberCandidate_holds. Separately the atlas settles the margin-sufficiency subquestion negatively: positive margin alone does not force a singleton fibre. The rest of part (b) — the radius constant, the singular classification, the graph-threshold program — is not transcribed and carries no verdict here." } }, "mais-issue-5-2026": { "citation": "Math for AI Safety (MAIS), issue #5, candidate negative resolution of MAIS-O7, filed 2026-08-03.", "locator": "https://github.com/lionellevine/MAIS/issues/5", "role": "work", "retrieved": "2026-09-05", "content_sha256": "edbaf2f2435c8854ce681a19cc2f5da10474e944fc2c9254624f0db50b5061fa", "notes": "Context for CONJ-027. The issue body points to the three-page candidate PDF MAIS-O07-candidate-solution.pdf, sha256 907ce2d0a5a11182995c4679acb6812109ecdb90e66da765a6f9037e5e9b59a8, archived alongside the issue #3 attachment in the shared literature tree after a 2026-09-06 audit found the graded artifact retained nowhere locally and reachable only through a GitHub attachment URL. The note gives the scalar M=N=r=1, H=2 counterexample: the rank-zero critical set is the origin with pair (1,1), while every point of the terminal fiber has pair (1/2,1), reversing the conjectured inequality. The note already quantifies over the terminal rung -- every point of C1 has pair (1/2,1) -- and states that the infimum is attained there; the rank-zero rung is a single point, so it is universal by construction. What the atlas adds is the certificate as one object: the exact exponent sets on both rungs, attainment as a proved proposition rather than an observation, and the whole of it for every positive target. The issue claims no human referee review. Issue bodies are editable in place, so an edit invalidates the body hash.", "mais_solution": { "problem": "MAIS-O7", "verdict": "CHECKED", "candidate_lean": "AISafetyAtlas.Conjectures.MAIS.O7CounterexampleAtEveryScale", "submitted": "A negative resolution: in the scalar instance the rank-zero critical set has local pair (1,1) while the terminal fibre has (1/2,1), reversing the conjectured strict increase.", "checked": "Confirmed and strengthened, though not in quantifier strength: the note already proves the pair at every point of the terminal rung and states that both infima are attained. o7CounterexampleAtEveryScale makes the two-rung certificate a single unconditional object -- the exact exponent set on each rung, attainment as a proposition, and every positive target. The conjectured increase is false. What was checked is the note's claim, not its argument: the note routes the rank-zero pair through the analytic Morse lemma at signature (2,2) and the atlas proves it by an explicit integral instead, so the note's own step is unverified. A 2026-09-06 audit added the two identifications the certificate rests on -- that its rank-zero rung is MAIS-A7's C_0 and its terminal rung that frame's exact-factorization fibre -- which had been transcriptions no check could see, and the descending-loss step that puts the instance inside print's stated setting." } }, "mais-issue-6-2026": { "citation": "Math for AI Safety (MAIS), issue #6, “full solution pending review” for MAIS-O23, filed 2026-08-04.", "locator": "https://github.com/lionellevine/MAIS/issues/6", "role": "work", "retrieved": "2026-08-19", "content_sha256": "4fd639c4322a3a3bd1b27fe6f14ee3de902961e0013485395e461d7cdc739a9b", "notes": "The construction the atlas checks: three models on two binary variables with a common behavioral transform. Open and unreviewed when this row was written. The issue credits the prior two-variable construction in MAIS issue #4 (MAIS-O34) by a different author and states its own contribution as making the MAIS-O23 consequence explicit across all three graphs. The atlas claims no priority and does not assert the issue's conclusion; what is recorded here is that a stated part of it machine-checks. Body sha256 as read 2026-08-20; issue bodies are editable in place, so an edit invalidates this grading.", "mais_solution": { "problem": "MAIS-O23", "verdict": "PARTIAL", "candidate_lean": "AISafetyAtlas.Examples.Causal.margin_class_not_identifiable", "submitted": "Filed as a full solution pending review: three models on two binary variables sharing a behavioural transform, making the MAIS-O23 consequence explicit across all three graphs.", "checked": "The construction machine-checks, and the atlas goes past it: a positive-dimensional colliding family rather than a point, a narrower two-graph reading that does not depend on the edgeless model being admissible, and transport to the source's real chart rather than a rational restriction. What is verified is narrower than what is claimed. A collision at one skeleton is a possibility result, not the general question, so the atlas does not endorse the full-solution framing; the printed question is answered separately and graded against the agenda. The issue credits its own construction to MAIS issue #4, and the atlas claims no priority for it." } }, "mais-issue-7-2026": { "citation": "Math for AI Safety (MAIS), issue #7, “MAIS-O24: candidate negative resolution — clauses (a) and (c) are incompatible as written”, filed 2026-08-04 by kumino.", "locator": "https://github.com/lionellevine/MAIS/issues/7", "role": "work", "retrieved": "2026-08-30", "content_sha256": "68e65b119a8923dd997e2ea75daea5331145706674a44ecc6d3f7c7b89a80ee7", "notes": "Context for CONJ-012. The candidate argues that prob:effective's clauses (a) and (c) cannot both hold, so the printed problem has no solution. Its argument lives in an attachment, MAIS-O24-candidate-solution.pdf, sha256 fd096dea004135fbbed054619857b307881d6432925d1ff1cae5ae3f45d16477, which grades MAIS-O24 at revision 43016a3e5c94edfca55ba49bd3e16770f7ac5dae; open-problems/MAIS-O24.md is byte-identical between that revision and the pin except for its status line. The atlas transcribes the incompatibility, repairs one circular step in the argument's choice of measure, and machine-checks the result as AISafetyAtlas.Examples.Causal.O24Refutation.not_o24_identifies_and_excluded; see docs/provenance/mais-o24-refutation.md. The issue claims no human referee review. Body sha256 as read 2026-08-30, updated_at 2026-08-04T02:32:33Z, no comments at that reading; issue bodies are editable in place, so an edit invalidates this record.", "mais_solution": { "problem": "MAIS-O24", "filed_by": "kumino", "verdict": "CHECKED", "candidate_lean": "AISafetyAtlas.Examples.Causal.O24Refutation.not_o24_identifies_and_excluded", "submitted": "A negative resolution: clauses (a) and (c) of the printed problem cannot both hold, so it has no solution.", "checked": "The incompatibility holds and is machine-checked. The argument as written had a circular step in its choice of measure, which the atlas repaired before checking rather than reproducing. AISafetyAtlas.Causal.O24Solution is the solution type with the proof obligations as fields, and it is empty." } }, "mais-issue-8-2026": { "citation": "Math for AI Safety (MAIS), issue #8, “MAIS-O31: candidate complete solution — generic chamber classification for one intervention in a binary chain”, filed 2026-08-04 by kumino.", "locator": "https://github.com/lionellevine/MAIS/issues/8", "role": "work", "retrieved": "2026-08-20", "content_sha256": "8e2e688eaac1a72f915aa787ad1e74676e6b72eff4f2796394e95b0a83fb8a96", "notes": "The candidate a conjecture row transcribes wherever that row is recorded. The issue derives four affine diagnostics and a generic chamber classification, and explicitly claims no human or journal referee verification. The transcribing row now quantifies directly over every threshold t in (0,1), exactly as the issue does. Separate lemmas derive and bound thresholds produced by margin-admissible utility gaps in the surrounding agenda problem; those lemmas do not restrict the conjecture. Body sha256 as read 2026-08-20; issue bodies are editable in place, so an edit invalidates this grading.", "mais_solution": { "problem": "MAIS-O31", "filed_by": "kumino", "verdict": "STATED_ONLY", "candidate_lean": "AISafetyAtlas.Conjectures.MAIS.maisO31_chainClassificationCandidate", "submitted": "A complete solution: four affine diagnostics and a generic chamber classification for one intervention in a binary chain.", "checked": "Transcribed at the issue's own quantifiers — every threshold in the open unit interval, all four binary local interventions, literal coordinate equality, and the issue's own scope exclusions as the chamber disjunct — and it is not proved. The embedding into the printed margin class is proved in both directions, so the statement is not left semantically detached. The atlas separately refutes the printed heuristic's claim that the endpoint marginal is recoverable, on an explicit box of Lebesgue measure 1/500." } }, "mais-issue-9-2026": { "citation": "Math for AI Safety (MAIS), issue #9, “MAIS-O33: candidate negative resolution — the persistent-corruption threshold is zero”, filed 2026-08-04 by kumino.", "locator": "https://github.com/lionellevine/MAIS/issues/9", "role": "work", "retrieved": "2026-08-30", "content_sha256": "9ff124f8cd8fb65d7393780384776f87a42a5870ce477b50da8dd9c315e9bd25", "notes": "Context for CONJ-023: the construction the atlas machine-checks. The body states the claim eta* = 0 and the argument lives in an attachment, MAIS-O33-candidate-solution.pdf (6 pages, MiKTeX pdfTeX-1.40.26), sha256 cf603983ca239bd933d5e3d0810ec5a660fbc1e587e2cce788c0e4edf12f45a6, read 2026-08-30. The note is audited in docs/provenance/mais-o33-statability.md, which found no error, and what the atlas added -- the transcription, the machine-check, and two changes of instance that each removed a dependency -- is in docs/provenance/mais-o33-refutation.md. The issue says its mathematics was generated by OpenAI Codex and offered under CC BY 4.0 with no human referee review claimed. Body sha256 as read 2026-08-30, updated_at 2026-08-04T02:33:16Z, no comments at that reading; issue bodies are editable in place, so an edit invalidates this record.", "mais_solution": { "problem": "MAIS-O33", "filed_by": "kumino", "verdict": "CHECKED", "candidate_lean": "AISafetyAtlas.Conjectures.MAIS.maisO33_etaStarIsZero", "submitted": "A negative resolution: the persistent-corruption threshold is zero.", "checked": "The submitted value is confirmed, with a caveat the ledger states rather than hides. The upper half is the proved content; the lower half follows from the sign clause together with the supremum of the empty set being zero, not from a proof that the printed zero endpoint is tolerable, and the graded correct answer is the baseline-relative form. An independent audit of the note found no error. The issue reports that its mathematics was generated by an AI system with no human referee review claimed." } }, "mais-o23-2026": { "citation": "Math for AI Safety (MAIS), open problem MAIS-O23, “Do margins imply behavioral identifiability of causal models?”, posed in agenda MAIS-A2 as Question 4.1; authored by Claude Fable 5 directed by Lionel Levine, audited by GPT 5.6 Sol.", "locator": "https://github.com/lionellevine/MAIS/blob/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/open-problems/MAIS-O23.md", "role": "work", "retrieved": "2026-08-19", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "content_sha256": "80a6aa32e387e12494e24c3c693cd2886aabe50a0889155affedfdf8e88cfdea", "notes": "The *question* only: if two models in A2's composite margin class have equal behavioral transforms, must they be equal? Not a definitional source. Local interventions, mixtures, CIDs, value and regret are RE24/Pearl/Everitt 2021. The six margins and the named transform family are A2 composites of those ingredients (strong-faithfulness move; RE24 mixing trick). Graded against mais-a2-2026's label q:ident, which fixes an arbitrary skeleton. The agenda is AI-written and not human-peer-reviewed." }, "mais-o38-2026": { "citation": "Math for AI Safety (MAIS), open problem MAIS-O38, “Polynomial sample complexity for dictionary uniqueness with growing sparsity”, posed in agenda MAIS-A3 as Problem 4.8; authored by Claude Fable 5 directed by Lionel Levine, audited by GPT 5.6 Sol.", "locator": "https://github.com/lionellevine/MAIS/blob/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/open-problems/MAIS-O38.md", "role": "work", "retrieved": "2026-08-27", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "content_sha256": "cea6784554724ce4e67b8967a9fe6fba6959e70a32494dc97b0b8fd3685c4d43", "notes": "Byte-for-byte the same problem statement as mais-a3-2026's prob:samples, so nothing is graded against this page that is not also in the agenda. It adds the throughout-convention the agenda leaves implicit — “a vector is k-sparse if at most k of its entries are nonzero” — and the literature framing: (k+1)*binom(m,k) generic samples suffice for a fixed A (Aharon-Elad-Bruckstein 2006, Hillar-Sommer 2015), Spielman-Wang-Wright 2012 control rival row sparsity for square dictionaries, and Awasthi-Vijayaraghavan 2018 give approximate recovery under restricted isometry and a triple-occurrence hypothesis. None of those results is transcribed by the atlas. The agenda is AI-written and not human-peer-reviewed." }, "mais-o7-2026": { "citation": "Math for AI Safety (MAIS), open problem MAIS-O7, “Monotonicity along the rank ladder”, posed in agenda MAIS-A7 as Problem 3.7.", "locator": "https://github.com/lionellevine/MAIS/blob/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/open-problems/MAIS-O7.md", "role": "work", "retrieved": "2026-09-05", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "content_sha256": "4b1add32acbf7d0e462e8b9d39144e5588a38cf446a1d57f7495ba183aaa3d8c", "notes": "One-page statement of the conjecture that the infimum of the two-sided local coefficient on C_k is nondecreasing in k, with attainment requested. Its mathematical statement is unchanged at the candidate revision 43016a3e5c94edfca55ba49bd3e16770f7ac5dae; only status links differ." }, "mais-o70-2026": { "citation": "Math for AI Safety (MAIS), open problem MAIS-O70, “Local coefficients on the template”, posed in agenda MAIS-A6 as prob:calibration.", "locator": "https://github.com/lionellevine/MAIS/blob/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/open-problems/MAIS-O70.md", "role": "work", "retrieved": "2026-09-03", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "content_sha256": "69a47687da280365ac2023ef4b69d571b472ccae5543251f00058d3683c7a47e", "notes": "One-page restatement pointing back into MAIS-A6, on the same footing the open-problems pages have for the other MAIS rows. It gives the loss in its printed Gaussian-expectation form, K(A,B) = (1/2) E_x ||(BA - B0A0)x||^2 with x ~ N(0, I_N); the Frobenius form is a consequence, not the definition, and the atlas derives it rather than substituting it. The agenda is AI-written and not human-peer-reviewed." }, "mais-o77-2026": { "citation": "Math for AI Safety (MAIS), open problem MAIS-O77, “Local pairs on the matrix-factorization fiber and every saddle set”, posed in agenda MAIS-A7 as Problem 3.9.", "locator": "https://github.com/lionellevine/MAIS/blob/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/open-problems/MAIS-O77.md", "role": "work", "retrieved": "2026-09-05", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "content_sha256": "07912e8c267f9b7866c12fe09a1277d5c880c0c9e60672c803be83a3b159a043", "notes": "One-page restatement of MAIS-A7 Problem 3.9. Part (a) concerns every point of the exact factorization fiber; part (b) concerns every point of every nonterminal saddle set C_k. Its mathematical statement is unchanged at the candidate revision 43016a3e5c94edfca55ba49bd3e16770f7ac5dae; only status links differ." }, "mais-open-problems-2026": { "citation": "Math for AI Safety (MAIS), master list of open problems — 92 numbered problems (MAIS-O1 … MAIS-O92), each with an AI-safety motivation, mathematical area, and difficulty rating, linked to a research agenda carrying full context.", "locator": "https://github.com/lionellevine/MAIS/tree/9dd29f8bf5ccd1e7701e300039b09ed4096b6516/open-problems", "role": "directory", "retrieved": "2026-08-07", "revision": "9dd29f8bf5ccd1e7701e300039b09ed4096b6516", "notes": "Problems are open to the registry's knowledge, literature-checked at release; MAIS-O60 was resolved in the negative in August 2026. Fourteen problems (MAIS-O1, O10-O22) concern bounded and quantitative Löb, cooperative AI, and program equilibrium — the area where the atlas already ships Löb, Gödel I/II and Tarski. Checked 2026-08-07: MAIS-O1 and MAIS-O16 are not statable against AISafetyAtlas.Logic.loeb, which is the unbounded theorem; both need proof-length metrics, arithmetized bounded provability, and expansion certificates that neither Mathlib nor Foundation provides. Checked 2026-08-30 against agenda A2: six of its fourteen printed problems cannot be stated in the atlas at all, and are recorded here rather than as conjecture-ledger rows. MAIS-O2 (prob:noisy) and MAIS-O35 (prob:starter-sample) need a sampled-action oracle -- the query layer observes exact real policy probabilities, so a transcript's answers are numbers rather than draws -- and O2 additionally needs independent response corruption at a known level, O35 a sampled-action minimax risk, regret adversaries, switching surfaces and binary-channel capacity. MAIS-O28 (prob:average) needs average-case admissibility against a declared continuous distribution on pairs of profiles; the atlas has worst-case admissibility only. MAIS-O29(c) (prob:boltzmann(c)) needs the risk under a fixed query law before a design problem over such laws can be posed; boltzmannMinimaxRisk already optimizes the query strategy away. MAIS-O30 (prob:restricted) needs a general Sigma_W construction, labels for edge and table functionals, and a combinatorial classifier in (G, W, O, Z). MAIS-O32 (prob:rate) has most of its vocabulary since the goal layer landed on 2026-08-30 and still lacks its own resolution radius over environments whose transitions read the action. The per-problem map, at declaration level, is docs/provenance/mais-a2-statement-coverage.md." }, "mathforaisafety-2026": { "citation": "AI Safety for Mathematicians, mathforaisafety.org — curated research directions for mathematicians entering AI safety (heuristic estimators, interpretability and feature manifolds, open-source game theory) plus a concrete-problems page.", "locator": "https://www.mathforaisafety.org/", "role": "directory", "retrieved": "2026-08-07", "notes": "Living directory; the site states the direction list is expanding and the writeups subject to change. Points at the AISI, Iliad, MAIS, and Timaeus directories also catalogued here." }, "pearl-2009-causality": { "citation": "J. Pearl, Causality: Models, Reasoning, and Inference, 2nd ed. Cambridge University Press, 2009.", "locator": "https://doi.org/10.1017/CBO9780511803161", "role": "work", "retrieved": "2026-08-19", "notes": "Ingredient for AISafetyAtlas.Causal.Model: causal Bayesian networks, the product factorization, and hard do(X=x) as truncated factorization. The atlas does not transcribe do-calculus or identification theory." }, "peleg-1998-effectivity-functions-rights": { "citation": "B. Peleg, “Effectivity functions, game forms, games, and rights,” Social Choice and Welfare, vol. 15, no. 1, pp. 67–80, 1998, doi: 10.1007/s003550050092.", "locator": "https://doi.org/10.1007/s003550050092", "role": "work", "retrieved": "2026-09-09", "notes": "Statement source for LAND-SOV-POWER-001. Graded statement by statement on 2026-09-11 in section 21 of docs/provenance/source-coverage-audit.md, and REGRADED TWICE ON 2026-09-20, first when the rights layer was built (17 Yes, 3 Partial, 7 No, 2 Beyond, up from 10 Yes, 1 Partial, 16 No, 2 Beyond) and again when print's other two societies were: 21 Yes, 1 Partial, 5 No, 2 Beyond. THE SECOND REGRADE closed the section's two remaining owed cells on one shared cost. smokeConstitution is Example 2.7, the office and the obligation not to smoke in front of a non-smoker; gibbardConstitution is Example 2.10, Gibbard's marriage society; and the eight gibbardInduced values with mem_gibbardConstitution_induced are Example 3.8's computation of (3.1) on the second. EXAMPLE 2.7 ANSWERS A QUESTION PRINT LEAVES OPEN: Definition 3.1's discussion says an EF derived from a constitution by (3.1) might not be superadditive and gives no example, and not_superadditive_officeConstitution is one at print's own society - a smoker alone may have the office smoky, a non-smoker alone may have it clear, and superadditivity would make the two of them effective for the empty set, which the obligation forbids - so not_represents_officeConstitution follows by Represents.superadditive. EXAMPLE 3.8's TWO VERIFICATIONS, which print leaves to the reader, are theorems: gibbardConstitution_upwardClosed and gibbardInduced_superadditive. Print's next sentence, that E is therefore representable, is NOT rendered: it uses the sufficiency half of Theorem 3.5, which this cluster does not have and which is costed on that row. A NOTATIONAL LOOSENESS RECURS: Examples 2.7 and 2.10 both write the standing assumption as gamma(S, empty) = A where page 69 writes {A}, the same singleton-for-set shorthand Example 2.3 uses. Harmless here, since Standing asks for the family and both constitutions satisfy it, and recorded because the Example 2.3 case shows the shorthand is not always harmless. Three of the four Partial rows closed the same day: not_forces_empty is condition (iii), effectivity_empty_eq_singleton_univ recovers condition (i) under print's surjectivity hypothesis, and Playable.mono_coalition is the superadditive-implies-coalition-monotone implication print left as 'straightforward'. Definition 3.4 remained Partial until 2026-09-20 and is now Yes and Wider; see THE RIGHTS LAYER below. PINNED as peleg1997.pdf, sha256 99039339aae01cb8e903ebab99e210d2ffdb7945cd9e89247abcdaf02976669a, manifested 2026-09-09, maintainer-supplied, 14 pp. YEAR: the filename and OpenAlex say 1997; the document's own first page says 'Soc Choice Welfare (1998) 15: 67-80' and 'Springer-Verlag 1998', received 25 Nov 1994, accepted 28 June 1996. The citation and the audit heading follow the document. File page p is journal page 66+p. Statements read from rendered pages 1, 2, 3, 4, 6, 7 and 8 = journal 67, 68, 69, 70, 72, 73 and 74; the mathematics is AMS-glyph and text extraction mangles it, so nothing is graded from extracted text. WHAT IS TAKEN: sections 2 and 3, i.e. the rights layer and the game-form layer, but not sections 4 and 5. Definition 3.2 is GameForm, Definition 3.3 is Forces and effectivity, Definition 3.4 is Represents, and the four conditions print imposes on an effectivity function - the whole-space condition, monotonicity w.r.t. the alternatives (3.3), monotonicity w.r.t. coalitions (3.5), and superadditivity (Definition 3.1) - are THEOREMS here about every game form rather than hypotheses. That is why Sovereignty.Separations can conclude that superadditivity carries no information about domination: within this class it cannot fail. ONE DEFECT OF PRINT, confirmed at the rendered page: Definition 3.3 on journal page 73 writes the alpha-effectivity family as sets B contained in S, where S is the coalition; it is a typo for B contained in A, the social states, as the same definition's opening line shows. NOT HERE: legality of a game form, Remark 2.5's equal-treatment condition, Examples 2.7, 2.10 and 3.8, the SUFFICIENCY direction of Theorem 3.5 (print's appendix construction of a game form from a prescribed superadditive effectivity function), section 4 on Gibbard's paradox, and section 5 on Sen's liberal paradox. THE RIGHTS LAYER, 2026-09-20. AISafetyAtlas.Sovereignty.Rights builds what this note previously listed as absent. Constitution is Definition 2.4, Constitution.induced is (3.1), Constitution.Standing is the standing assumption on journal page 69, MonotoneAssign / MonotoneAlternatives / MonotoneRights / MonotoneCoalitions are (3.2) to (3.5), Superadditive is Definition 3.1 on a bare effectivity function, and Represents is Definition 3.4 against the constitution rather than against a second game form. Surjectivity of the outcome function is still not a standing assumption and is now DERIVED where print assumes it: Represents.surjective_outcome. With it, Playability.effectivity_univ_eq_nonempty is print's EF condition (ii), the last of the four that had no rendering. Theorem 3.5 is Partial rather than No: Represents.superadditive is necessity of its condition (ii); necessity of (i) as printed is not available, because (3.3) quantifies gamma over every set of rights while a representation constrains only the diagonal gamma(S, alpha(S)), which print's own page 70 says is all that enters the analysis of a society at a given date. AISafetyAtlas.Examples.Sovereignty.represented_not_monotoneAlternatives exhibits the shortfall. A SECOND LOOSENESS OF PRINT, and it is looseness rather than a typo: Example 2.3 lists gamma(1, rho_1) and gamma(2, rho_1) as two sets each - the minimal attainable sets - while writing gamma(N, rho_1) = 2^A minus the empty set in full, in the same sentence. So within one example one coalition's family is given entire and two are given by generators; Remark 2.2 on the same page supplies the reading and Example 2.10 writes the closure out with its B-plus notation. Taken as whole families the listing fails print's own (3.3), and AISafetyAtlas.Examples.Sovereignty.not_represents_of_induced_eq_listed shows no game form whatever represents such a constitution - which would contradict Example 3.6. Relatedly, print's E(empty) = A on pages 72 and 73 is not a typo either but print's own abbreviation, declared on page 69: 'henceforth, we shall denote a singleton {a} by a'. WITNESSES: Examples.Sovereignty.Rights renders Example 2.3 as shirtConstitution, Example 3.6 as represents_shirtGame and Example 3.7 as not_represents_sequentialGame." }, "power-sovereignty-proposal-unpublished": { "citation": "“A Formal System of Power, Sovereignty, and Cognitive Sovereignty,” unpublished note, 2026. The document names no author and carries no venue; it declares itself a proposed operational theory whose definitions are modeling choices.", "role": "work", "retrieved": "2026-09-12", "notes": "Held privately; there is no URL, which is why no locator is recorded. Statement source for LAND-SOV-SERVICE-001, LAND-SOV-QUANT-001, LAND-SOV-TRANSFER-001, LAND-SOV-CATALOGUE-001 and LAND-SOV-AUDIT-001. UNPUBLISHED, so it is deliberately NOT graded in docs/provenance/source-coverage-audit.md and no row claims coverage of it. Pinned by sha256 with its two companion files in docs/provenance/formal-power-proposal-triage.md, which also triages all 64 of its numbered results. sha256 2ee1d90530cc43d2cb9eeb065cc9e6f5f78d5d44e9a727ac25989aa011f17584. The accompanying finite_models_power.py is its executable model and is search step 2 evidence, not a formalization the atlas depends on. The Lean draft PowerKernel_uncompiled.lean that section 12 names is not in the folder and was never available." }, "richens-everitt-2024": { "citation": "J. Richens and T. Everitt, “Robust agents learn causal world models,” International Conference on Learning Representations (ICLR), 2024, arXiv:2402.10877.", "locator": "https://arxiv.org/abs/2402.10877", "role": "work", "retrieved": "2026-08-19", "notes": "Primary peer-reviewed source for the causal/decision primitives: Def. 2 local intervention v |-> f(v) and the displayed mechanism-replacement formula; Def. 3 mixtures; Def. 4 single-decision CID; Assumps. 1–2 unmediated task and domain dependence; section 2.2 expected utility and regret; section 2.3 masking Pa_D -> Pa_D' as a local intervention. Theorems 1–2 (Lebesgue-almost-every identifiability) are not formalized. Causal.Model transcribes the finite categorical specialization of Definitions 2–3, over an ordered field, and the mechanism-replacement formula; Causal.Decision formalizes the unmediated finite decision layer. They are not transcribed from MAIS-A2." }, "schumacher-vose-whitley-2001": { "citation": "C. Schumacher, M. D. Vose, and L. D. Whitley, “The No Free Lunch and Problem Description Length,” in Proc. Genetic and Evolutionary Computation Conference (GECCO 2001), Morgan Kaufmann, 2001, pp. 565–570.", "locator": "https://dl.acm.org/doi/10.5555/2955239.2955325", "role": "work" }, "survey-ref-001": { "survey_number": 1, "citation": "K. Gödel, “Über formal unentscheidbare Sätze der Principia Mathematica und verwandter Systeme I,” Monatshefte Für Math. Phys., vol. 38, no. 1, pp. 173–198, Dec. 1931, doi: 10.1007/BF01700692.", "locator": "https://doi.org/10.1007/BF01700692", "role": "work" }, "survey-ref-002": { "survey_number": 2, "citation": "A. M. Turing, “On Computable Numbers, with an Application to the Entscheidungsproblem,” Proc. Lond. Math. Soc., vol. s2-42, no. 1, pp. 230–265, 1937, doi: 10.1112/plms/s2-42.1.230.", "locator": "https://doi.org/10.1112/plms/s2-42.1.230", "role": "work" }, "survey-ref-005": { "survey_number": 5, "citation": "D. H. Wolpert, “Physical limits of inference,” Phys. Nonlinear Phenom., vol. 237, no. 9, pp. 1257–1281, Jul. 2008, doi: 10.1016/j.physd.2008.03.040.", "locator": "https://doi.org/10.1016/j.physd.2008.03.040", "role": "work" }, "survey-ref-008": { "survey_number": 8, "citation": "R. V. Yampolskiy, “Uncontrollability of Artificial Intelligence,” presented at the IJCAI-21 Workshop on Artificial Intelligence Safety (AISafety2021), Montreal, Quebec, Canada, Aug. 2021. Accessed: Jul. 01, 2021. [Online]. Available: http://arxiv.org/abs/2008.04071", "locator": "http://arxiv.org/abs/2008.04071", "role": "work", "notes": "HELD 2026-09-13 as yampolskiy-arxiv-v1-2020-on-controllability-of-ai.pdf, sha256 57abe0f8bc5d5eb30d4bad0da76cf48eb3b2c399474f50c0d54b4ed3c4233eaf, 59 pp. ITS OWN TITLE IS 'On Controllability of AI', not the title this entry cites. Page 32 read from a rendered page image: the only thing in the paper labelled a theorem is a block quotation of Alfonseca et al.'s Theorem 1, with their proof and corollary. Print states no result of its own." }, "survey-ref-018": { "survey_number": 18, "citation": "D. H. Wolpert, “The Existence of A Priori Distinctions Between Learning Algorithms,” Neural Comput., vol. 8, no. 7, pp. 1391–1420, Oct. 1996, doi: 10.1162/neco.1996.8.7.1391.", "locator": "https://doi.org/10.1162/neco.1996.8.7.1391", "role": "work" }, "survey-ref-019": { "survey_number": 19, "citation": "D. H. Wolpert and W. G. Macready, “No free lunch theorems for optimization,” IEEE Trans. Evol. Comput., vol. 1, no. 1, pp. 67–82, Apr. 1997, doi: 10.1109/4235.585893.", "locator": "https://doi.org/10.1109/4235.585893", "role": "work" }, "survey-ref-021": { "survey_number": 21, "citation": "J. Klamka, “Uncontrollability and unobservability of multivariable systems,” IEEE Trans. Autom. Control, vol. 17, no. 5, pp. 725–726, Oct. 1972, doi: 10.1109/TAC.1972.1100128.", "locator": "https://doi.org/10.1109/TAC.1972.1100128", "role": "work", "retrieved": "2026-09-11", "content_sha256": "fdaa652dc63e69a5bcae88cd679d2b7832d7e07fea0bf2c1a7acb18f6f0a2751", "notes": "Statement source of BY-001 and BY-002. PINNED 2026-09-11 as klamka1972.pdf in the literature directory, supplied by the maintainer after IEEE Xplore declined; manifested 2026-09-11. Both pages read as rendered images and graded statement by statement in section 25 of docs/provenance/source-coverage-audit.md, which opened at 0 Yes, 1 Partial, 16 No, 4 Beyond and stands at 7 Yes, 4 Partial, 6 No, 7 Beyond as of 2026-09-20. CONTENT: a two-page note on the linear time-invariant system xdot = Ax + Bu, y = Cx giving SUFFICIENT conditions for uncontrollability and unobservability by counting Jordan blocks from the minimal polynomial - Theorem 1, Theorem 2 and four corollaries, all one-directional. It states neither the Kalman rank criterion nor the Hautus test, and quotes the Jordan-row criterion of Chen and Desoer as its engine." }, "survey-ref-022": { "survey_number": 22, "citation": "A. P. Guerreiro, C. M. Fonseca, and L. Paquete, “The Hypervolume Indicator: Problems and Algorithms,” ArXiv200500515 Cs, May 2020, Accessed: May 08, 2021. [Online]. Available: http://arxiv.org/abs/2005.00515", "locator": "http://arxiv.org/abs/2005.00515", "role": "work" }, "survey-ref-023": { "survey_number": 23, "citation": "R. M. Karp, “Reducibility among combinatorial problems,” in Complexity of computer computations, Springer, 1972, pp. 85–103, doi: 10.1007/978-1-4684-2001-2_9.", "locator": "https://doi.org/10.1007/978-1-4684-2001-2_9", "role": "work" }, "survey-ref-024": { "survey_number": 24, "citation": "J. Klamka, “Controllability of dynamical systems. A survey,” Bulletin of the Polish Academy of Sciences: Technical Sciences, vol. 61, no. 2, pp. 335–342, 2013, doi: 10.2478/bpasts-2013-0031.", "locator": "https://doi.org/10.2478/bpasts-2013-0031", "role": "work" }, "survey-ref-025": { "survey_number": 25, "citation": "R. C. CONANT and W. R. ASHBY, “Every good regulator of a system must be a model of that system,” Int. J. Syst. Sci., vol. 1, no. 2, pp. 89–97, Oct. 1970, doi: 10.1080/00207727008920220.", "locator": "https://doi.org/10.1080/00207727008920220", "role": "work", "notes": "Held 2026-09-13 privately, sha256 fba0430f1196748e81915a11860948b5dbb249e861b9a086953b67420362ec7d, manifested 2026-09-13. The face reads 'Int. J. Systems Sci., 1970, vol. 1, No. 2, 89-97'. Journal page 96 read from a rendered page image; print's theorem and its one-element lemma are quoted in BY-003's statability note." }, "survey-ref-026": { "survey_number": 26, "citation": "R. W. Ashby, Introduction to Cybernetics.1961 Edition. Chapman & Hall, 1961.", "locator": "https://archive.org/details/introductiontocy0000ashb", "role": "work" }, "survey-ref-027": { "survey_number": 27, "citation": "H. Touchette and S. Lloyd, “Information-theoretic approach to the study of control systems,” Phys. Stat. Mech. Its Appl., vol. 331, no. 1, pp. 140–172, Jan. 2004, doi: 10.1016/j.physa.2003.09.007.", "locator": "https://doi.org/10.1016/j.physa.2003.09.007", "role": "work" }, "survey-ref-028": { "survey_number": 28, "citation": "J. McDowell, “Virtue and Reason,” The Monist, vol. 62, no. 3, pp. 331–350, Jul. 1979, doi: 10.5840/monist197962319.", "locator": "https://doi.org/10.5840/monist197962319", "role": "work" }, "survey-ref-029": { "survey_number": 29, "citation": "S. McKeever and M. Ridge, “The Many Moral Particularisms,” Can. J. Philos., vol. 35, no. 1, pp. 83–106, 2005, doi: 10.1080/00455091.2005.10716582.", "locator": "https://doi.org/10.1080/00455091.2005.10716582", "role": "work" }, "survey-ref-030": { "survey_number": 30, "citation": "P. S.-H. Tsu, “Can Virtue Be Codified?: An Inquiry on the Basis of Four Conceptions of Virtue,” in Virtue’s Reasons, Routledge, 2017, doi: 10.4324/9781315314259-5.", "locator": "https://doi.org/10.4324/9781315314259-5", "role": "work" }, "survey-ref-031": { "survey_number": 31, "citation": "K. J. Arrow, “A Difficulty in the Concept of Social Welfare,” J. Polit. Econ., vol. 58, no. 4, pp. 328–346, Aug. 1950, doi: 10.1086/256963.", "locator": "https://doi.org/10.1086/256963", "role": "work" }, "survey-ref-032": { "survey_number": 32, "citation": "G. Arrhenius, “The Impossibility of a Satisfactory Population Ethics,” in Descriptive and Normative Approaches to Human Behavior, vol. Volume 3, 0 vols., WORLD SCIENTIFIC, 2011, pp. 1–26. doi: 10.1142/9789814368018_0001.", "locator": "https://doi.org/10.1142/9789814368018_0001", "role": "work" }, "survey-ref-033": { "survey_number": 33, "citation": "P. Eckersley, “Impossibility and Uncertainty Theorems in AI Value Alignment (or why your AGI should not have a utility function),” ArXiv190100064 Cs, Mar. 2019, Accessed: Jun. 28, 2021. [Online]. Available: http://arxiv.org/abs/1901.00064", "locator": "http://arxiv.org/abs/1901.00064", "role": "work", "notes": "HELD 2026-09-13 as eckersley-arxiv-v1-2019-impossibility-and-uncertainty-theorems-in-ai-value-alignment.pdf, sha256 8a3c7680398e75e0f70147e642be2919cb298f5da0040ec9019837a7cf78937a, 13 pp. VERSION: the banner on the face reads arXiv:1901.00064v3, 5 Mar 2019. Read 2026-09-13: the paper states no numbered result of its own and transforms other people's impossibility theorems, chiefly Arrow's and Arrhenius's." }, "survey-ref-034": { "survey_number": 34, "citation": "J. Kleinberg, S. Mullainathan, and M. Raghavan, “Inherent Trade-Offs in the Fair Determination of Risk Scores,” ArXiv160905807 Cs Stat, Nov. 2016, Accessed: Jun. 28, 2021. [Online]. Available: http://arxiv.org/abs/1609.05807", "locator": "http://arxiv.org/abs/1609.05807", "role": "work" }, "survey-ref-035": { "survey_number": 35, "citation": "K. K. Saravanakumar, “The Impossibility Theorem of Machine Fairness -- A Causal Perspective,” ArXiv200706024 Cs Stat, Jan. 2021, Accessed: Aug. 12, 2021. [Online]. Available: http://arxiv.org/abs/2007.06024", "locator": "http://arxiv.org/abs/2007.06024", "role": "work" }, "survey-ref-036": { "survey_number": 36, "citation": "S. Armstrong and S. Mindermann, “Occam’s razor is insufficient to infer the preferences of irrational agents,” Adv. Neural Inf. Process. Syst., vol. 31, 2018, Accessed: Jun. 28, 2021. [Online]. Available: https://proceedings.neurips.cc/paper/2018/hash/d89a66c7c80a29b1bdbab0f2a1a94af8-Abstract.html", "locator": "https://proceedings.neurips.cc/paper/2018/hash/d89a66c7c80a29b1bdbab0f2a1a94af8-Abstract.html", "role": "work", "retrieved": "2026-09-11", "notes": "Statement source for BY-011 (the whole AISafetyAtlas.Preference cluster, six modules) and for LAND-PREF-KNOW-001. Graded statement by statement for the first time on 2026-09-11, in section 20 of docs/provenance/source-coverage-audit.md, which opened at 30 Yes, 4 Partial, 10 No, 2 Beyond and stands at 31 Yes, 3 Partial, 10 No, 2 Beyond as of 2026-09-20, when appendix B.1's first branch was added: print writes two things about inaction two paragraphs apart under different planners, the atlas had only the second, and IsRationalPlanner with regret_noop_eq_zero_of_rationalPlanner is the first, with regret_noop_eq_zero_or_overrides proving the two exhaustive rather than merely consistent. PINNED 2026-09-11, and it had never been pinned before: the cluster carried registry rows, a landmark row and a 25-row statement map in docs/provenance/a1-a3-b1-b3-b7-statement-maps.md while no copy of the paper existed in the private literature store at all. Manifested 2026-09-11. THREE FILES. Publisher main PDF, sha256 7e580c07450226dc9891c3120d94817df16a46291783f0318c9764b9e2f80375, 12 pp., from https://proceedings.neurips.cc/paper_files/paper/2018/file/d89a66c7c80a29b1bdbab0f2a1a94af8-Paper.pdf; NeurIPS's own metadata gives pp. 5598-5609 of Advances in Neural Information Processing Systems vol. 31. Publisher supplemental appendix, sha256 7da3a4b24f8d5f2b2f33684dcee793b3ec2df09426c12bb87aba27f29ff22872, 5 pp., the single file Occam_appendix.pdf inside the Supplemental.zip at the same URL stem. arXiv:1712.05812v6, sha256 0cc5ff7a514580ec57744d024f7b6b8108271be85e731127309f5056aede1341, 17 pp., support only. WHERE THE STATEMENTS ARE: the main PDF carries Theorem 1, Theorem 2, Propositions 3, 4, 7, 8, Definition 5, Lemma 6 and Conjecture 9; it does NOT carry Proposition 10 or Definition 11, which are in the supplemental appendix - so AISafetyAtlas.Preference.Override formalizes from the supplemental. The appendix's own page numbers run 13-17, continuing the main text, and arXiv v6 is exactly the two together. The numbering agrees across all three; no concordance table is needed. AUTHOR ORDER: the publisher's typeset page 1 prints Soeren Mindermann first and Stuart Armstrong second, while arXiv v6 prints Armstrong first; both carry the footnote 'Equal contribution', so neither order is a precedence claim. NeurIPS's own metadata record lists Armstrong first, which is what this citation follows. ONE LEDGER CORRECTION: AISafetyAtlas.Preference.Regret was listed as owed against 'Everitt et al., IJCAI 2017, Theorem 11 certificate'. That is the module's hypothesis; what it states is the closing sentence of this paper's section 4.1.2, which is a section heading and not a numbered statement. Print itself attributes the half-maximal regret inequality to Everitt et al. 2017 rather than proving it, so the atlas's split - a HalfMaximalRegretBound hypothesis discharged in the wireheading cluster - is print's own structure. NOT COVERED: the MDP/R dynamics, appendix B's doubled-environment construction, the qualitative section 6, appendix C, and the time-bounded complexity measures of appendix A." }, "survey-ref-037": { "survey_number": 37, "citation": "H. G. Rice, “Classes of Recursively Enumerable Sets and Their Decision Problems,” Trans. Am. Math. Soc., vol. 74, no. 2, pp. 358–366, 1953, doi: 10.2307/1990888.", "locator": "https://doi.org/10.2307/1990888", "role": "work" }, "survey-ref-038": { "survey_number": 38, "citation": "A. Church, “An Unsolvable Problem of Elementary Number Theory,” Am. J. Math., vol. 58, no. 2, pp. 345–363, 1936, doi: 10.2307/2371045.", "locator": "https://doi.org/10.2307/2371045", "role": "work" }, "survey-ref-039": { "survey_number": 39, "citation": "G. J. Chaitin, Information, Randomness And Incompleteness: Papers On Algorithmic Information Theory. 1987, doi: 10.1142/0531.", "locator": "https://doi.org/10.1142/0531", "role": "work" }, "survey-ref-040": { "survey_number": 40, "citation": "A. Tarski, “The Concept of Truth in Formalized Languages,” in Logic, Semantics, Metamathematics, A. Tarski, Ed. Oxford University Press, 1936, pp. 152–278.", "locator": "https://archive.org/details/logicsemanticsme0000tars", "role": "work" }, "survey-ref-041": { "survey_number": 41, "citation": "O. B. Bassler, “The Surveyability of Mathematical Proof: A Historical Perspective,” Synthese, vol. 148, no. 1, pp. 99–133, 2006, doi: 10.1007/s11229-004-6221-7.", "locator": "https://doi.org/10.1007/s11229-004-6221-7", "role": "work" }, "survey-ref-042": { "survey_number": 42, "citation": "S. Ben-David, P. Hrubeš, S. Moran, A. Shpilka, and A. Yehudayoff, “Learnability can be undecidable,” Nat. Mach. Intell., vol. 1, no. 1, pp. 44–48, Jan. 2019, doi: 10.1038/s42256-018-0002-3.", "locator": "https://doi.org/10.1038/s42256-018-0002-3", "role": "work" }, "survey-ref-043": { "survey_number": 43, "citation": "L. Reyzin, “Unprovability comes to machine learning,” Nature, vol. 565, no. 7738, Art. no. 7738, Jan. 2019, doi: 10.1038/d41586-019-00012-4.", "locator": "https://doi.org/10.1038/d41586-019-00012-4", "role": "work" }, "survey-ref-044": { "survey_number": 44, "citation": "L. G. Valiant, “A theory of the learnable,” Commun. ACM, vol. 27, no. 11, pp. 1134–1142, Nov. 1984, doi: 10.1145/1968.1972.", "locator": "https://doi.org/10.1145/1968.1972", "role": "work", "notes": "HELD 2026-09-13 as valiant-cacm-1984-a-theory-of-the-learnable.pdf, sha256 5ab696a5f2afe34fc184872a11e7fe120782094b3b4810a0f1ac0a21f99ece59, 9 pp., the CACM research-contributions typesetting. Carries Claims 1 to 4 and no other numbered statement. Inventory only; no statement read against the tree." }, "survey-ref-045": { "survey_number": 45, "citation": "D. Foster and H. P. Young, “On the Impossibility of Predicting the Behavior of Rational Agents,” The Johns Hopkins University,Department of Economics, 423, Jun. 2001. Accessed: Jul. 01, 2021. [Online]. Available: https://ideas.repec.org/p/jhu/papers/423.html", "locator": "https://ideas.repec.org/p/jhu/papers/423.html", "role": "work" }, "survey-ref-046": { "survey_number": 46, "citation": "R. Koppl and J. B. R. Jr, “All That I Have to Say Has Already Crossed Your Mind,” Metroeconomica, vol. 53, no. 4, pp. 339–360, 2002, doi: 10.1111/1467-999X.00147.", "locator": "https://doi.org/10.1111/1467-999X.00147", "role": "work" }, "survey-ref-047": { "survey_number": 47, "citation": "A. Auger and O. Teytaud, “Continuous Lunches Are Free Plus the Design of Optimal Optimization Algorithms,” Algorithmica, vol. 57, no. 1, pp. 121–146, May 2010, doi: 10.1007/s00453-008-9244-5.", "locator": "https://doi.org/10.1007/s00453-008-9244-5", "role": "work" }, "survey-ref-048": { "survey_number": 48, "citation": "D. H. Wolpert and W. G. Macready, “Coevolutionary free lunches,” IEEE Trans. Evol. Comput., vol. 9, no. 6, pp. 721–735, Dec. 2005, doi: 10.1109/TEVC.2005.856205.", "locator": "https://doi.org/10.1109/TEVC.2005.856205", "role": "work" }, "survey-ref-049": { "survey_number": 49, "citation": "A. Hyvärinen and P. Pajunen, “Nonlinear independent component analysis: Existence and uniqueness results,” Neural Netw., vol. 12, no. 3, pp. 429–439, Apr. 1999, doi: 10.1016/S0893-6080(98)00140-3.", "locator": "https://doi.org/10.1016/S0893-6080(98)00140-3", "role": "work" }, "survey-ref-050": { "survey_number": 50, "citation": "J. Peters, D. Janzing, and B. Schölkopf, Elements of Causal Inference: Foundations and Learning Algorithms. Cambridge, MA, USA: MIT Press, 2017.", "locator": "https://people.math.ethz.ch/~jopeters/elements.html", "role": "work", "notes": "HELD 2026-09-13 as peters-janzing-scholkopf-mit-press-2017-elements-of-causal-inference.pdf, sha256 52f8843f135c969f31fd93d52ac380d96ac9cdbd53c5df2214b02f80d4455598, 289 pp., the MIT Press open-access deposit. NOT READ." }, "survey-ref-051": { "survey_number": 51, "citation": "F. Locatello et al., “Challenging Common Assumptions in the Unsupervised Learning of Disentangled Representations,” 2019. Accessed: Jun. 29, 2021. [Online]. Available: http://proceedings.mlr.press/v97/locatello19a.html", "locator": "http://proceedings.mlr.press/v97/locatello19a.html", "role": "work", "notes": "HELD 2026-09-13 as locatello-etal-pmlr-2019-challenging-common-assumptions-in-the-unsupervised-learning-of-disentangled-representations.pdf, sha256 bbce8bcfc6e25f92c9e74d9cc9cbdd5039688a8cb7d572d544f9dd63e533a9ef, 11 pp., PMLR v97 camera-ready. Theorem 1 read 2026-09-13 from a rendered image of page 3 and quoted in BY-023's statability note." }, "survey-ref-052": { "survey_number": 52, "citation": "I. I. Bojinov and G. Basse, “A General Theory of Identification,” Harvard Business School, Feb. 2020. Accessed: Jun. 11, 2021. [Online]. Available: https://www.hbs.edu/faculty/Pages/item.aspx?num=57688", "locator": "https://www.hbs.edu/faculty/Pages/item.aspx?num=57688", "role": "work" }, "survey-ref-053": { "survey_number": 53, "citation": "D. H. Wolpert, “Computational capabilities of physical systems,” Phys. Rev. E, vol. 65, no. 1, p. 016128, Dec. 2001, doi: 10.1103/PhysRevE.65.016128.", "locator": "https://doi.org/10.1103/PhysRevE.65.016128", "role": "work" }, "survey-ref-054": { "survey_number": 54, "citation": "D. Wolpert, “Constraints on physical reality arising from a formalization of knowledge,” ArXiv171103499 Phys., Jun. 2018, Accessed: Jun. 22, 2021. [Online]. Available: http://arxiv.org/abs/1711.03499", "locator": "http://arxiv.org/abs/1711.03499", "role": "work", "notes": "VERSION: no peer-reviewed published version exists; checked 2026-08-15, the arXiv record carries no journal-ref and no DOI and no journal version is indexed. The preprint is therefore canonical by necessity, and the atlas reads v3, the latest of three versions. This is the one atlas source where a preprint is the authority." }, "survey-ref-055": { "survey_number": 55, "citation": "A. Devereaux, R. Koppl, S. Kauffman, and A. Roli, “Constraints on modeling systems from within systems: the principle of frame relativity,” Social Science Research Network, Rochester, NY, SSRN Scholarly Paper ID 3968077, Nov. 2021. doi: 10.2139/ssrn.3968077.", "locator": "https://doi.org/10.2139/ssrn.3968077", "role": "work" }, "survey-ref-056": { "survey_number": 56, "citation": "M. Alfonseca, M. Cebrian, A. F. Anta, L. Coviello, A. Abeliuk, and I. Rahwan, “Superintelligence cannot be contained: Lessons from Computability Theory,” J. Artif. Intell. Res., vol. 70, pp. 65–76, Jan. 2021, doi: 10.1613/jair.1.12202.", "locator": "https://doi.org/10.1613/jair.1.12202", "role": "work", "notes": "HELD 2026-09-13 in its PUBLISHED form as alfonseca-cebrian-fernandez-anta-coviello-abeliuk-rahwan-jair-2021-superintelligence-cannot-be-contained.pdf, sha256 ee8cfd460b9074e085142f84ddbfee03606a55954f7e5125e9f7d79c39f0b1a4; running head 'Journal of Artificial Intelligence Research 70 (2021) 65-76'. Journal page 71 read from a rendered page image: Theorem 1, the harming problem is undecidable, by reduction from halting; Assumption 2; Corollary 3, the containment problem is incomputable." }, "survey-ref-057": { "survey_number": 57, "citation": "R. Carey, “Incorrigibility in the CIRL Framework,” ArXiv170906275 Cs, Jun. 2018, Accessed: Jul. 01, 2021. [Online]. Available: http://arxiv.org/abs/1709.06275", "locator": "http://arxiv.org/abs/1709.06275", "role": "work", "notes": "HELD 2026-09-13 as carey-arxiv-v2-2018-incorrigibility-in-the-cirl-framework.pdf, sha256 696cb676dd2a0525ba4a4a637e937c6dabef984d85a9805c76241d1f191c25c2, 9 pp., banner arXiv:1709.06275v2. Carries Definition 1 and Theorem 1. Inventory only; no statement read against the tree." }, "survey-ref-058": { "survey_number": 58, "citation": "L. Orseau and S. Armstrong, “Safely interruptible agents,” in Proceedings of the Thirty-Second Conference on Uncertainty in Artificial Intelligence, Arlington, Virginia, USA, Jun. 2016, pp. 557–566.", "locator": "https://www.auai.org/uai2016/proceedings/papers/68.pdf", "role": "work", "notes": "HELD 2026-09-13 as orseau-armstrong-uai-2016-safely-interruptible-agents.pdf, sha256 bfb4209e0c3f517dea29003b6ccb69dfb10a28897949f2912c4c7e79b3d0607e, 10 pp., from the UAI 2016 proceedings server. Carries Definitions 1 to 19, Lemmas 3 to 24, Theorems 7, 8, 14, 15, 17 and 18, and Proposition 11. Inventory only; no statement read against the tree." }, "survey-ref-059": { "survey_number": 59, "citation": "D. Hadfield-Menell, A. Dragan, P. Abbeel, and S. Russell, “The off-switch game,” in Proceedings of the 26th International Joint Conference on Artificial Intelligence, Melbourne, Australia, Aug. 2017, pp. 220–227, doi: 10.24963/ijcai.2017/32.", "locator": "https://doi.org/10.24963/ijcai.2017/32", "role": "work", "notes": "HELD 2026-09-13 as hadfield-menell-dragan-abbeel-russell-arxiv-2017-the-off-switch-game.pdf, sha256 2e8131110d1921e8de0a096ea3f39a2012be684532ba58db0105186e0eae0f7d, 8 pp., banner arXiv:1611.08219v3. Carries Theorems 1 and 2 and Corollary 1. Inventory only; no statement read against the tree." }, "survey-ref-060": { "survey_number": 60, "citation": "T. Wängberg, M. Böörs, E. Catt, T. Everitt, and M. Hutter, “A Game-Theoretic Analysis of the Off-Switch Game,” in Artificial General Intelligence, Cham, 2017, pp. 167–177. doi: 10.1007/978-3-319-63703-7_16.", "locator": "https://doi.org/10.1007/978-3-319-63703-7_16", "role": "work" }, "survey-ref-061": { "survey_number": 61, "citation": "E. M. E. Mhamdi, R. Guerraoui, H. Hendrikx, and A. Maurer, “Dynamic safe interruptibility for decentralized multi-agent reinforcement learning,” in Proceedings of the 31st International Conference on Neural Information Processing Systems, Red Hook, NY, USA, Dec. 2017, pp. 129–139.", "locator": "https://proceedings.neurips.cc/paper_files/paper/2017/hash/812b4ba287f5ee0bc9d43bbf5bbe87fb-Abstract.html", "role": "work" }, "survey-ref-062": { "survey_number": 62, "citation": "M. H. Löb, “Solution of a Problem of Leon Henkin,” J. Symb. Log., vol. 20, no. 2, pp. 115–118, 1955, doi: 10.2307/2266895.", "locator": "https://doi.org/10.2307/2266895", "role": "work" }, "survey-ref-063": { "survey_number": 63, "citation": "V. Vinge, “TECHNOLOGICAL SINGULARITY,” presented at the VISION-21 Symposium sponsored by NASA Lewis Research Center and the Ohio Aerospace Institute, 1993. Accessed: Jul. 01, 2021. [Online]. Available: https://frc.ri.cmu.edu/~hpm/book98/com.ch1/vinge.singularity.html", "locator": "https://frc.ri.cmu.edu/~hpm/book98/com.ch1/vinge.singularity.html", "role": "work" }, "survey-ref-064": { "survey_number": 64, "citation": "R. V. Yampolskiy, “Unpredictability of AI: On the Impossibility of Accurately Predicting All Actions of a Smarter Agent,” J. Artif. Intell. Conscious., vol. 07, no. 01, pp. 109–118, Mar. 2020, doi: 10.1142/S2705078520500034.", "locator": "https://doi.org/10.1142/S2705078520500034", "role": "work" }, "survey-ref-065": { "survey_number": 65, "citation": "R. V. Yampolskiy, “Unexplainability and Incomprehensibility of AI,” J. Artif. Intell. Conscious., vol. 07, no. 02, pp. 277–291, Sep. 2020, doi: 10.1142/S2705078520500150.", "locator": "https://doi.org/10.1142/S2705078520500150", "role": "work", "notes": "PREPRINT HELD 2026-09-13 as yampolskiy-arxiv-v1-2019-unexplainability-and-incomprehensibility-of-ai.pdf, sha256 17d210ef6ed75966ac63b84b960bcf611276742b9112fb22d2770f9c7bd2c319, 14 pp., dated June 20 2019 on its own face. THE PUBLISHED JAIC TEXT THIS ENTRY CITES IS NOT HELD and the two have NOT been compared. Read 2026-09-13: the preprint states no theorem, lemma, proposition or corollary of its own and cites other people's, Charlesworth's in particular." }, "survey-ref-066": { "survey_number": 66, "citation": "A. Charlesworth, “Comprehending software correctness implies comprehending an intelligence-related limitation,” ACM Trans. Comput. Log., vol. 7, no. 3, pp. 590–612, Jul. 2006, doi: 10.1145/1149114.1149119.", "locator": "https://doi.org/10.1145/1149114.1149119", "role": "work" }, "survey-ref-067": { "survey_number": 67, "citation": "J. Hernandez-orallo, “A formal definition of intelligence based on an intensional variant of Kolmogorov complexity,” in In Proceedings of the International Symposium of Engineering of Intelligent Systems (EIS’98, 1998, pp. 146–163.", "locator": "https://dmip.webs.upv.es/papers/EIS98.pdf", "role": "work" }, "survey-ref-068": { "survey_number": 68, "citation": "R. V. Yampolskiy, “What are the ultimate limits to computational techniques: verifier theory and unverifiability,” Phys. Scr., vol. 92, no. 9, p. 093001, Jul. 2017, doi: 10.1088/1402-4896/aa7ca8.", "locator": "https://doi.org/10.1088/1402-4896/aa7ca8", "role": "work", "notes": "PREPRINT HELD 2026-09-13 as yampolskiy-arxiv-v1-2016-verifier-theory-and-unverifiability.pdf, sha256 d07efd1c4fd3ece1e91a0c377cb100db35fe7fd4ca158e7da03e839ea54215c7, 13 pp. ITS OWN TITLE IS 'Verifier Theory and Unverifiability', which is not the Physica Scripta title this entry cites; THE PUBLISHED TEXT IS NOT HELD and the two have NOT been compared. Read 2026-09-13: no numbered result and no prose theorem; print's abstract describes a research programme." }, "survey-ref-069": { "survey_number": 69, "citation": "J. van Leeuwen and J. Wiedermann, “Impossibility Results for the Online Verification of Ethical and Legal Behaviour of Robots,” Utrecht University, Utrecht, UU-PCS-2021-02, 2021. Accessed: Aug. 12, 2021. [Online]. Available: http://www.cs.uu.nl/groups/AD/UU-PCS-2021-02.pdf", "locator": "https://web.archive.org/web/20220222045551/http://www.cs.uu.nl/groups/AD/UU-PCS-2021-02.pdf", "role": "work", "notes": "VERSION: the citation is the survey’s and is preserved as such; it is an unrefereed Utrecht technical report, and it is what Verification.Robot transcribes. A peer-reviewed counterpart exists - Wiedermann and van Leeuwen, Validating Non-trivial Semantic Properties of Autonomous Robots, in Philosophy and Theory of Artificial Intelligence 2021, SAPERE vol. 63, pp. 91-104, Springer 2022, doi 10.1007/978-3-031-09153-7_8 - and the two REUSE THE SAME NUMBERS FOR DIFFERENT STATEMENTS. Report Theorem 1 says no algorithmic procedure decides whether a robot actions always satisfy a non-trivial P; chapter Theorem 1 says that for all REGULAR P, P is trivial if and only if R_P is recursive, strengthened in the following remark to not even recursively enumerable. The corollaries and the Theorem 2s differ the same way, and the chapter adds a regularity hypothesis the report Theorem 1 does not carry. So action_safety_unverifiable matches the report Theorem 1 and Corollary 1 and only the easy half of the chapter Theorem 1. Both copies were read locally on 2026-08-15." }, "survey-ref-070": { "survey_number": 70, "citation": "C. J. Reynolds, “On the Computational Complexity of Action Evaluations,” Netherlands, 2005. Accessed: Jul. 01, 2021. [Online]. Available: https://www.media.mit.edu/publications/on-the-computational-complexity-of-action-evaluations/", "locator": "https://www.media.mit.edu/publications/on-the-computational-complexity-of-action-evaluations/", "role": "work" }, "survey-ref-071": { "survey_number": 71, "citation": "M. Brundage, “Limitations and risks of machine ethics,” J. Exp. Theor. Artif. Intell., vol. 26, no. 3, pp. 355–372, Jul. 2014, doi: 10.1080/0952813X.2014.895108.", "locator": "https://doi.org/10.1080/0952813X.2014.895108", "role": "work", "notes": "HELD 2026-09-13 as brundage-jetai-2014-limitations-and-risks-of-machine-ethics.pdf, sha256 b0914881b7dfa613b3742a6a46b9d7e3dd8c9c6bc521142f23e9c33bc454499e, 20 pp., the author's own deposit. Read 2026-09-13: no numbered statement of any kind; it is a critical essay and does not carry BY-034's complexity claim." }, "survey-ref-072": { "survey_number": 72, "citation": "H. W. Lin, M. Tegmark, and D. Rolnick, “Why Does Deep and Cheap Learning Work So Well?,” J. Stat. Phys., vol. 168, no. 6, pp. 1223–1247, Sep. 2017, doi: 10.1007/s10955-017-1836-5.", "locator": "https://doi.org/10.1007/s10955-017-1836-5", "role": "work" }, "survey-ref-073": { "survey_number": 73, "citation": "C. S. Calude, S. Heidari, and J. Sifakis, “What Neural Networks Are (Not) Good For?,” Research report CDMTCS- 556, Aug. 2021.", "locator": "https://www.cs.auckland.ac.nz/research/groups/CDMTCS/researchreports/publication-archive.php?selected-id=556", "role": "work", "notes": "The survey cites the 2021 technical report. The locator links to the official University of Auckland CDMTCS Research Report 556 archive page. A later journal publication is C. S. Calude, S. Heidari, and J. Sifakis, 'What perceptron neural networks are (not) good for?', Information Sciences, vol. 621, 2023, pp. 844–857, doi: 10.1016/j.ins.2022.11.083. Statement correspondence between the technical report and journal publication has been verified by the contributor; the technical-report citation is retained to preserve the survey's source reference." }, "survey-ref-074": { "survey_number": 74, "citation": "C. A. E. Goodhart, “Problems of Monetary Management: The UK Experience,” in Monetary Theory and Practice: The UK Experience, C. A. E. Goodhart, Ed. London: Macmillan Education UK, 1984, pp. 91–121. doi: 10.1007/978-1-349-17295-5_4.", "locator": "https://doi.org/10.1007/978-1-349-17295-5_4", "role": "work" }, "survey-ref-075": { "survey_number": 75, "citation": "M. Strathern, “‘Improving ratings’: audit in the British University system,” Eur. Rev., vol. 5, no. 3, pp. 305–321, Jul. 1997, doi: 10.1002/(SICI)1234-981X(199707)5:3<305::AID-EURO184>3.0.CO;2-4.", "locator": "https://doi.org/10.1002/(SICI)1234-981X(199707)5:3<305::AID-EURO184>3.0.CO;2-4", "role": "work" }, "survey-ref-076": { "survey_number": 76, "citation": "D. T. Campbell, “Assessing the impact of planned social change,” Eval. Program Plann., vol. 2, no. 1, pp. 67–90, Jan. 1979, doi: 10.1016/0149-7189(79)90048-X.", "locator": "https://doi.org/10.1016/0149-7189(79)90048-X", "role": "work" }, "survey-ref-077": { "survey_number": 77, "citation": "T. Everitt, V. Krakovna, L. Orseau, M. Hutter, and S. Legg, “Reinforcement Learning with a Corrupted Reward Channel,” ArXiv170508417 Cs Stat, Aug. 2017, Accessed: May 13, 2021. [Online]. Available: http://arxiv.org/abs/1705.08417", "locator": "http://arxiv.org/abs/1705.08417", "role": "work" }, "survey-ref-078": { "survey_number": 78, "citation": "W. J. Howe and R. V. Yampolskiy, “Impossibility of Unambiguous Communication as a Source of Failure in AI Systems,” presented at the IJCAI-21 Workshop on Artificial Intelligence Safety (AISafety2021), Montreal, Quebec, Canada, Aug. 2021. doi: 10.13140/RG.2.2.13245.28641.", "locator": "https://doi.org/10.13140/RG.2.2.13245.28641", "role": "work" }, "timaeus-projects-2026": { "citation": "Timaeus project ideas — a directory of project ideas in developmental interpretability and singular learning theory.", "locator": "https://timaeus.co/projects", "role": "directory", "retrieved": "2026-08-07", "notes": "Reached via mathforaisafety-2026." }, "touchette-lloyd-2003-preprint": { "citation": "H. Touchette and S. Lloyd, “Information-theoretic approach to the study of control systems,” arXiv:physics/0104007v2, 16 May 2003. Preprint of Phys. Stat. Mech. Its Appl. 331(1):140–172, 2004.", "locator": "https://arxiv.org/abs/physics/0104007v2", "role": "work" }, "uhler-etal-2013-faithfulness": { "citation": "C. Uhler, G. Raskutti, P. Bühlmann, and B. Yu, “Geometry of the faithfulness assumption in causal inference,” Annals of Statistics, vol. 41, no. 2, pp. 436–463, 2013, doi: 10.1214/12-AOS1080.", "locator": "https://doi.org/10.1214/12-AOS1080", "role": "work", "retrieved": "2026-08-19", "notes": "Ingredient for the *role* of (M1)–(M6), not for the six inequalities themselves: strong faithfulness replaces a measure-zero unfaithfulness assumption by an explicit margin, and shows the excluded set can be large. MAIS-A2 cites this analogy; the atlas records it as the literature ground of that move. The six numbered conditions remain an A2 composite." }, "wolpert-1996-lack": { "citation": "D. H. Wolpert, “The Lack of A Priori Distinctions Between Learning Algorithms,” Neural Comput., vol. 8, no. 7, pp. 1341–1390, Oct. 1996, doi: 10.1162/neco.1996.8.7.1341.", "locator": "https://doi.org/10.1162/neco.1996.8.7.1341", "role": "work" }, "yanofsky-2003-lawvere": { "citation": "N. S. Yanofsky, “A Universal Approach to Self-Referential Paradoxes, Incompleteness and Fixed Points,” Bulletin of Symbolic Logic, vol. 9, no. 3, pp. 362–386, 2003, doi: 10.2178/bsl/1058448677.", "locator": "https://doi.org/10.2178/bsl/1058448677", "role": "work", "notes": "Statement source for the sets-and-functions form exposed as CLM-LAWVERE-001; Lawvere's 1969 cartesian-closed-category theorem is strictly broader than the Atlas declaration." }, "governance-kernel-sketch-unpublished": { "citation": "“Governance, agency, and cognitive sovereignty: an Atlas integration specification,” unpublished sketch, 2026. The title is the document's own first line; it names no author and carries no venue. It was produced by a language model with access only to the public branch of this repository, and its section 0 declares its own contents a specification sketch.", "role": "work", "retrieved": "2026-09-12", "notes": "Held privately; there is no URL, which is why no locator is recorded. sha256 419a02dd7a24a0d28e49434310cf610673b448b21e56aaaf90d142efe905d695, pinned with its two companion files in docs/provenance/governance-sketch-uniform-decision.md. Statement source for LAND-KNOW-UNIFORM-001 and nothing else. UNPUBLISHED, so it is deliberately NOT graded in docs/provenance/source-coverage-audit.md and no row claims coverage of it. Its companion COVERAGE.md and INTEGRATION.md describe fourteen Lean files, forty-two proof scripts and 6,091 executed checks that are NOT in the directory; that evidence class is treated as absent, and every grading against this source is against the prose of FORMAL_SPECIFICATION.md alone." }, "lin-wonham-1988-observability": { "citation": "F. Lin and W. M. Wonham, “On observability of discrete-event systems,” Information Sciences, vol. 44, no. 3, pp. 173–198, 1988.", "locator": "https://doi.org/10.1016/0020-0255(88)90001-1", "role": "work", "retrieved": "2026-09-13", "notes": "Held privately, sha256 73030c40c4fcda09c7876e27ecbd95565509dbd2e04e3e622ecc3bc6b49be9d3, manifested 2026-09-13. Read 2026-09-13 at folios 177-181 from rendered page images. Volume, pages and year are on the face at p.173; publisher line is Elsevier Science Publishing Co., Inc. 1988. This is the PUBLISHED ANCHOR for the pairwise half of LAND-KNOW-UNIFORM-001: p.177 defines observability as ker P <= act_K, a pairwise relation; p.178 says act_K is only a tolerance relation, reflexive and symmetric and NOT transitive; Theorem 2.1 at p.181 is nonetheless an iff. The mechanism is at p.179, where the supervisor is any psi : Sigma x X -> {0,1} separating two unions, one binary decision per event. The atlas does not reproduce the discrete-event setting and claims no coverage of this paper; it takes only the reason the pairwise condition is complete there." }, "keiding-1985-stability-effectivity": { "citation": "H. Keiding, “Necessary and sufficient conditions for stability of effectivity functions,” International Journal of Game Theory, vol. 14, no. 2, pp. 93–101, 1985.", "locator": "https://doi.org/10.1007/BF01770226", "role": "work", "retrieved": "2026-09-12", "notes": "Held privately, sha256 f39fab6dd1c7a5d38035d122b84f7ac58819062542fb56b1b97aa6d47ace1082, manifested 2026-09-12. Volume, issue and pages are on the face. Read 2026-09-12 and re-read 2026-09-13 at folios 96-97 from rendered page images. Definition 3.3 (p.96) is the cycle; Remark 3.4 says its 'or' is not exclusive; Lemma 3.5 relates strong cycles to cycles; Theorem 3.6 (p.97) cycle implies unstable; Lemma 3.7 extends an acyclic relation on a FINITE set to an ordering; Theorem 3.8 acyclic implies stable. Statement source for LAND-SOV-STABILITY-001, which transcribes Definition 3.3 only. The two theorems are NOT formalized. Corrected 2026-09-16: the previous note said the reason was that this repository has no preference-profile type, which was false -- SocialChoice.Profile is Fin N -> Preorder' A, orderings one per player, which is K(A)^N. What is missing is the core C(A,E,R^N), an outcome no coalition can block; Sovereignty.Domination blocks a demand and mentions no preference, and TrulyPlayable.nonmonotonicCore is a different object. So the two theorems want a definition and two proofs, plus finiteness for Lemma 3.7." }, "grossi-gabbay-van-der-torre-2010-norm-implementation": { "citation": "D. Grossi, D. Gabbay, and L. van der Torre, “The Norm Implementation Problem in Normative Multi-Agent Systems,” in M. Dastani, K. V. Hindriks, and J.-J. Ch. Meyer (eds.), Specification and Verification of Multi-agent Systems, Springer, 2010, pp. 195–224.", "locator": "https://doi.org/10.1007/978-1-4419-6984-2_7", "role": "work", "retrieved": "2026-09-13", "notes": "Held privately as the whole containing volume, sha256 b076b12eec92a07901f9ee486ff2ad2370bdc00d06d4e99781599a86c005ec94, manifested 2026-09-13 section 6. Born-digital, not a scan. The chapter's own first page prints DOI 10.1007/978-1-4419-6984-2_7, so the identification is established on the face rather than inferred from a metadata record. Read 2026-09-13 at folios 195, 212 and 216 from rendered page images. Statement source for the regiment and enforce pair in LAND-SOV-DEONTIC-001. Folio 212, section 7.4.1: regimentation is a model update that DELETES transitions, R_a^{m'} := R_a^m minus the pairs whose precondition holds and whose target is a violation, after which it becomes impossible to execute that transition. Folio 216, section 7.5: perfect enforcement modifies payoffs and is defined by NOT moving the transitions, its printed conditions opening W = W', W_end = W'_end and {R_a} = {R'_a}. The chapter's extensive-game framework, its retarded preconditions at folio 216, and its enforcer and sanction sections are NOT formalized." }, "jones-sergot-1996-institutionalised-power": { "citation": "A. J. I. Jones and M. Sergot, “A Formal Characterisation of Institutionalised Power,” Journal of the IGPL, vol. 4, no. 3, pp. 427–443, 1996.", "locator": "https://doi.org/10.1093/jigpal/4.3.427", "role": "work", "retrieved": "2026-09-12", "notes": "Held privately, sha256 e5ebe993c8571a4ed75899c40b716a37491eeb9ab1305379d24674320521117b, manifested 2026-09-12. An ABBYY OCR scan, so statements are read from rendered page images only. Read 2026-09-12 at folios 431-435 and 2026-09-13 at folio 427. The running foot 'J. of the IGPL, Vol. 4 No. 3, pp. 427-443 1996' is legible at 155 dpi INCLUDING the year; an earlier note in the manifest called the year token OCR rubble and that was a rendering artefact of the lower resolution, corrected 2026-09-13. Statement source for LAND-SOV-DEONTIC-001, which takes ONLY the abstract's separation claim at p.427: 'Following a lead from jurisprudential discussions of legal power, we distinguish institutionalised power from permission and practical possibility.' The paper's conditional connective, its minimal-model semantics at p.435 and its rejection of RCM and RI at p.433 and PTR at p.435 are NOT formalized." }, "wooldridge-van-der-hoek-2005-obligations-normative-ability": { "citation": "M. Wooldridge and W. van der Hoek, “On obligations and normative ability: Towards a logical analysis of the social contract,” Journal of Applied Logic, vol. 3, no. 3–4, pp. 396–420, 2005, doi: 10.1016/j.jal.2005.04.006.", "locator": "https://doi.org/10.1016/j.jal.2005.04.006", "role": "work", "retrieved": "2026-09-13", "notes": "Held as wooldridge-van-der-hoek-published-jal-2005-on-obligations-and-normative-ability.pdf, sha256 21970069a6e93508c54cdf3b8632ebb54cf151a39e865d42e4f67e32aa7b705f, manifested 2026-09-13, 25 pp. Every citation field above is from the document face: the running foot prints the journal, volume, year and the span 396-420, and the PDF title field is the doi. Statements read 2026-09-13 from rendered page images at journal pages 405, 407, 408, 409 and 410. WHY IT IS CATALOGUED: Agotnes, van der Hoek and Wooldridge 2007 name it as the source for the obligation notion their own paper does not carry, and it turned out to carry the published statement of a claim this repository had built from an unpublished sketch - Proposition 3 item 1 is the NATL* form of the sketch's GK3. Graded in section 27 of docs/provenance/source-coverage-audit.md." }, "reuel-bucknall-2025-open-problems-technical-ai-governance": { "citation": "A. Reuel, B. Bucknall, et al., “Open Problems in Technical AI Governance,” Transactions on Machine Learning Research, 04/2025; arXiv:2407.14981v2.", "locator": "https://arxiv.org/abs/2407.14981", "role": "directory", "retrieved": "2026-09-13", "notes": "Held privately, sha256 926848a53d65da04..., manifested 2026-09-13. Ninety-nine numbered open problems, read in full by layout extraction 2026-09-13; the section prose behind them is not read. A DIRECTORY and not a work: it states questions, not theorems, so nothing here is graded against it and it is not in the coverage audit. It is cited by LAND-AUDIT-LAG-001 as the source of the question that bridge answers." }, "bengio-2026-international-ai-safety-report": { "citation": "Y. Bengio (Chair) et al., “International AI Safety Report 2026,” February 2026; Expert Advisory Panel nominated by 30+ countries and international organisations.", "locator": null, "role": "directory", "retrieved": "2026-09-13", "notes": "Held privately, sha256 e2a35f439cafade360e4df4a2f91f217a16174bd1231f97e64f7208ffa93e225, 220 pages, manifested 2026-09-13. Title, chair and panel read from the document. READ EXTENT: the contributor page, the section map, the key-information blocks of section 2.2.2 (Loss of control) and section 2.3.2 (Risks to human autonomy), and page 90 in full. The rest is NOT read. Section 2.3.2 begins on page 89; the two sentences LAND-SOV-ASSESSMENT-001 quotes were verified on 2026-09-14 against a rendered image of page 90, not against extracted text. Note a discrepancy internal to the report: the key-information block on page 89 says 'approximately 6% lower following several months of exposure to AI-assisted diagnosis' while the body text on page 90 says 'three months after the introduction of AI support'; the row quotes the page 90 form. A DIRECTORY and not a work: it surveys evidence and states no theorem, so nothing is graded against it and it is not in the coverage audit. Its own caveat -- 'research into the relationship between use of AI and cognitive offloading and critical thinking is nascent, and further studies supporting these findings are warranted' -- is part of any claim citing it." }, "danlyng-econlib-2026": { "citation": "D. Lyng, Econlib -- a Lean 4 + Mathlib formalization of economic and political theory, commit 003655ccf010cdf44c4f67d6675167b54ce0e9df, 2026. Apache-2.0.", "locator": "https://github.com/danlyng/Econlib", "role": "work", "retrieved": "2026-09-16", "notes": "FORMALIZATION SOURCE, NOT STATEMENT SOURCE. AISafetyAtlas.Analysis.Blackwell is adapted from this repository's Econlib/Math/Analysis/Blackwell.lean (sha256 1879097686a79081481812eab72871a33ac01d635ef49369759a602b333f6c9d); the atlas file header carries the upstream file, its hash and the atlas changes, and AISafetyAtlas/Upstream/LICENSE-NOTICE carries the repository-level notice. Only that file was taken. The seventeen-file Optimization/DynamicProgramming/ subtree was read and DECLINED, because its three carriers are deterministic over an arbitrary state type, stochastic over Fin n, and measure-valued over R, while AISafetyAtlas.Decision.MDP is stochastic over an arbitrary state type -- Econlib has stochastic, or unconstrained, never both. Costing: docs/agent/policy/lean-reuse-sources.md. Upstream toolchain leanprover/lean4 v4.29.0. No upstream NOTICE file, so redistribution incurs Apache-2.0 4(a)-(c) and not 4(d)." }, "sidhu-etal-2026-open-problems-ai-incident-governance": { "citation": "H. Sidhu, C. Scholefield, K. Annan, J. Hernandez, W. Nieh Hou, A. Alshaikhi, K. Chin, P. Gipiskis, “Open Problems in AI Incident Governance,” arXiv:2607.05163v1 [cs.CY], 6 July 2026.", "locator": "https://arxiv.org/abs/2607.05163", "role": "directory", "retrieved": "2026-09-13", "content_sha256": "14578ecc350c891b133a0a6df446613afb50355b03c010f8b362fda1c46d4002", "notes": "A DIRECTORY: it poses open problems about incident governance and states no theorem, so nothing is graded against it and it is not in the coverage audit. The question LAND-INCIDENT-COUNT-001 answers is its individuation question -- when a single incident begins and ends, and when harm traced to one model counts as one incident or many. Read for that question only; the surrounding policy prose is not read." }, "skalse-2022-defining-characterizing-reward-hacking": { "citation": "J. Skalse, N. H. R. Howe, D. Krasheninnikov, and D. Krueger, “Defining and Characterizing Reward Hacking,” in Advances in Neural Information Processing Systems 35 (NeurIPS 2022), 2022.", "locator": "https://proceedings.neurips.cc/paper_files/paper/2022/hash/3d719fee332caa23d5038b8a90e81796-Abstract-Conference.html", "role": "work", "retrieved": "2026-09-10", "notes": "Statement source for LAND-GOODHART-HACKABILITY-001. Graded in section 26 of docs/provenance/source-coverage-audit.md, first on 2026-09-13 at 0 Yes, 1 Partial, 15 No and regraded on 2026-09-20 at 5 Yes, 0 Partial, 13 No when the occupancy layer was built. PINNED as skalse-howe-krasheninnikov-krueger-published-neurips-2022-defining-and-characterizing-reward-hacking.pdf, sha256 634ffa7ccb0225296482ef2961a38ba175bd8c1b97b998556a6bcfa7ad560210, 12 pp, manifested 2026-09-10. The published NeurIPS version is canonical; arXiv v2, sha256 d9a8567f..., carries the same Definitions 1-4, Lemma 1, Theorems 1-3 and Corollaries 1-3 identical in wording and numbering plus an appendix with the proofs and Propositions 1-4. No version fork. Statements read from rendered pages 4 to 8. WHAT IS TAKEN: section 4 only - the setup of section 4.1 (discounted visit counts, the two embeddings, value as a linear functional of the reward) and all of section 4.2 (Definitions 1 and 2, the equivalent/trivial side conditions, symmetry, non-transitivity, and footnotes 4 and 5). ONE THING PRINT DEFINES THAT IS NOT HERE: print defines J by the expected discounted return along a trajectory and then observes it equals the pairing with the visit counts; the atlas defines it by the pairing and does not prove the identity. Its own row in section 26 costs it. ONE LOOSENESS OF PRINT, recorded because it changes a statement: footnote 5 says a trivial reward is a simplification of ANY reward, but Definition 2's own non-degeneracy clause asks for a distinction to lose, so the base must be non-trivial; Goodhart.simplifies_of_trivial carries that hypothesis and Goodhart.not_simplifies_of_trivial_base shows it cannot be dropped. NOT HERE: Lemma 1 and Theorems 1, 2 and 3 - the geometry, which needs a topology on policy space or a rank computation in occupancy space - together with Definitions 3 and 4, Corollaries 1 to 3, the worked computations of sections 5.2 and 5.3, and the arXiv-only appendix. EXTERNAL: audieleon/goodhart carries Skalse.skalse_theorem1, skalse_theorem3 and skalse_corollary3 in Lean, with policies as occupancy vectors given as data; it was adjudicated on 2026-09-10 in docs/provenance/external-formalizations.md with the verdict cite, do not vendor, and version skew unresolved. That verdict is undisturbed: its substrate takes F as given where this derives it from the transition kernel, the initial distribution and the discount." } }, "results": [ { "id": "BY-001", "name": "Unobservability", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "A system's internal state cannot in general be reconstructed from its observable outputs.", "original_source_refs": [ "survey-ref-021" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "module": "AISafetyAtlas.LinearSystems.Flow", "build_command": "lake build AISafetyAtlas.LinearSystems.Flow AISafetyAtlas.Examples.LinearSystems.Flow", "declaration": "AISafetyAtlas.LinearSystems.determinesStateSolOn_iff_isObservable", "relationship": "EXACT", "content_source_refs": [ "survey-ref-021" ], "scope_delta": { "summary": "The survey row asserts that a system's internal state cannot in general be reconstructed from its observable outputs. AISafetyAtlas.LinearSystems.Dynamics states that property - DeterminesStateOn: two trajectories under the same input that agree on the output agree on the state - and not_determinesStateOn_of_not_isObservable proves it FAILS whenever the Kalman observability condition does, with blind_not_determinesStateOn a named system where it fails. determinesStateOn_iff_isObservable is the full characterization, which page 726 of the cited source states and attributes to Chen and Desoer. THE CLASS IS THE SOURCE'S OWN, AND IS STATED HERE BECAUSE THE ROW'S SENTENCE DOES NOT CARRY IT: what is proved is the finite-dimensional linear time-invariant system x' = Ax + Bu, y = Cx over the complex numbers, which is exactly the object of survey-ref-021 (Klamka 1972) and not dynamical systems at large. WIDER THAN THE SOURCE ON ONE AXIS: the time window is a parameter and the equivalence holds on every non-empty open one, where print writes the system on the whole line. AT THE SOURCE'S SOLUTION CLASS (2026-10-05): the graded declaration is determinesStateSolOn_iff_isObservable, stated over IsSolution, the integral (Caratheodory) form x t = x 0 + integral of (Ax + Bu), which admits inputs with jumps -- the piecewise-continuous class print imports from Chen and Desoer (a step input is witnessed, integrator_step_isSolution; that the whole class gives solutions of this kind is standard and not proved here); integrator_step_isSolution and integrator_step_not_isTrajectory show a step input that is a solution and not a classical run. The classical-solution form determinesStateOn_iff_isObservable, on which the closure audit found the sufficiency side narrower than print, is kept beside it. pair_not_determinesStateSolOn is a non-degenerate witness (a real readout that misses a state). Graded statement by statement in section 25 of docs/provenance/source-coverage-audit.md; NC-013 in docs/provenance/formalization-search.json records that the six-corpus sweep found the continuous-time equivalence in no proof assistant on 2026-09-20.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.LinearSystems.determinesStateSolOn_iff_isObservable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.LinearSystems.DeterminesStateSolOn", "AISafetyAtlas.LinearSystems.IsSolution", "AISafetyAtlas.LinearSystems.outputSignal", "AISafetyAtlas.LinearSystems.IsObservable" ], "application": "Control: the outputs of a linear time-invariant system determine the state it passed through, on any non-empty open time window, EXACTLY WHEN the Kalman observability condition holds, at the source's solution class (integral-form solutions, inputs with jumps admitted). The survey row's claim is the failing direction of this equivalence." }, { "atlas_declaration": "AISafetyAtlas.Examples.LinearSystems.pair_not_determinesStateSolOn", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Examples.LinearSystems.pairA", "AISafetyAtlas.Examples.LinearSystems.firstB", "AISafetyAtlas.Examples.LinearSystems.firstC" ], "application": "Control: a named system with a real readout whose outputs still do not determine its state, among solutions with arbitrary locally integrable inputs." }, { "atlas_declaration": "AISafetyAtlas.LinearSystems.determinesStateOn_iff_isObservable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.LinearSystems.DeterminesStateOn", "AISafetyAtlas.LinearSystems.IsTrajectoryOn", "AISafetyAtlas.LinearSystems.outputSignal", "AISafetyAtlas.LinearSystems.IsObservable" ], "application": "Control: the outputs of a linear time-invariant system determine the state it passed through, on any non-empty open time window, EXACTLY WHEN the Kalman observability condition holds. The survey row's claim is the failing direction of this equivalence." }, { "atlas_declaration": "AISafetyAtlas.LinearSystems.not_determinesStateOn_of_not_isObservable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.LinearSystems.eigenTrajectory", "AISafetyAtlas.LinearSystems.unobservableSubspace" ], "application": "Control: when the Kalman condition fails, two runs under the same input agree on every output and differ in state. This is the survey row's sentence as a theorem." }, { "atlas_declaration": "AISafetyAtlas.Examples.LinearSystems.blind_not_determinesStateOn", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Examples.LinearSystems.integratorA", "AISafetyAtlas.Examples.LinearSystems.blindC" ], "application": "Control: a named system whose state its outputs do not determine. The row's claim at a witness rather than in prose." } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "A3 (2026-07-19): AFP 'observability' keyword hits (FSM_Tests Observability, protocol refinement, etc.) are DISTINCT from Klamka multivariable control-theoretic unobservability (survey-ref-021). Candidate formalizations cleared. See docs/provenance/a3-by001-unobservability-triage.md. RETIRED STATABILITY RECORD (2026-09-21). This row carried a statability verdict until it carried Lean; policy rejects a verdict on a row that has Lean, so the record is kept here rather than dropped. Its verdict was TRIAGED_DISTINCT and it remains correct about what it was about - the EXTERNAL candidate, AnandGokhale/LeanForControl, whose adapted layer is algebraic and is still not coverage of this row. What covers the row is atlas-original and was built on 2026-09-20. The note read: RETRIAGED 2026-09-13, from CANDIDATE_LEAD. Both of the two reasons the 2026-09-11 note gave for stopping at a lead are now discharged. THE SOURCE IS READ: Klamka's note was supplied by the maintainer and both printed pages were read on 2026-09-11 from rendered images, and section 25 of docs/provenance/source-coverage-audit.md grades it statement by statement. THE CANDIDATE IS ADJUDICATED AND FOUND DISTINCT: AnandGokhale/LeanForControl's LinearSystems track is REPRODUCED and in-tree as AISafetyAtlas.LinearSystems under LAND-LINSYS-001, and what it carries is the Kalman rank criteria and the Hautus tests - neither of which is any statement in print. WHAT WAS OWED AND IS NOW BUILT (2026-09-13): the previous note said the row is owed print's numbered results. They are in the tree, at print's own antecedent, in AISafetyAtlas.LinearSystems.BlockBound - not_isControllable_of_ceil_div_gt_rank and ..._gt_width, not_isObservable_of_ceil_div_gt_rank and ..._gt_height - reached through two lemmas Mathlib does not have, finrank_ker_comp_le and finrank_ker_pow_le, which are print's counting step with no Jordan form in it. THE TRAJECTORY BRIDGE IS BUILT AND THAT SENTENCE IS RETRACTED. The bridge landed 2026-09-20 and this note was swept 2026-09-21, one day late; the same one-day lag is recorded in section 25 of the coverage audit, which had already spent a week asserting a blocker it had itself removed. Until the sweep this note read: 'WHAT IS STILL NOT COVERED, AND IS WHY A FRESH FORMALIZATION REMAINS OWED: none of this defines a trajectory, a solution or an output signal ... and it is a substrate gap, not a candidate gap.' Every clause of it is now false. AISafetyAtlas.LinearSystems.Dynamics defines IsTrajectoryOn - a solution of x' = Ax + Bu asked on a set of times - and outputSignal, which is y = Cx; AISafetyAtlas.LinearSystems.Flow builds the matrix exponential, variation of constants and the adjoint argument. Print's two properties are in the tree AS PROPERTIES and each is EQUIVALENT to the algebraic criterion named for it, which is the equivalence page 726 states and attributes to Chen and Desoer. THE OBSERVABILITY SIDE, WHICH IS THIS ROW. DeterminesStateOn says that two trajectories under the same input agreeing on the output agree on the state, and determinesStateOn_iff_isObservable proves it equivalent to the Kalman rank condition on ANY non-empty open time window. AISafetyAtlas.Examples.LinearSystems.blind_not_determinesStateOn exhibits a system whose state its outputs do not determine, which is this row's informal claim at a witness rather than in prose. The controllability side is on BY-002. PROMOTED 2026-09-21, on the maintainer's decision. This note recorded the promotion as open, because moving a survey row to covered moves the headline coverage counts. The row now carries lean_artifact and a formalization record graded EXACT. The class restriction the row's own sentence does not carry is stated in the scope_delta rather than left implicit: what is proved is the finite-dimensional linear time-invariant system over the complex numbers, which is the object of survey-ref-021 and not dynamical systems at large. THE IDENTIFICATION THAT WAS ASSUMED IS NOW PROVED (2026-09-20): print's index is the multiplicity in the minimal polynomial and the exponent in the new statements was characterized by stabilization of the generalized eigenspace chain. The two are equal; print cites Zadeh and Desoer for it and does not prove it, and maxGenEigenspaceIndex_eq_rootMultiplicity_minpoly in AISafetyAtlas.LinearSystems.BlockBound now does, over an arbitrary field and with no Jordan form. Print's numbered results are available with print's own nu computed from A. Print's Theorem 2 and Corollaries 2 to 4 are the observability side and are the ones this row is stated against. PREVIOUS NOTE, KEPT: Re-triaged 2026-09-11. AnandGokhale/LeanForControl formalizes the Kalman and Hautus observability criteria in Lean; its LinearSystems track is now in-tree as AISafetyAtlas.LinearSystems under LAND-LINSYS-001. That is a CANDIDATE_LEAD and not coverage, for two independent reasons: the ported layer is algebraic and proves nothing about reconstructing a state from output trajectories, and Klamka's 1972 note - the row's cited source - is described in catalogue as giving a minimal-polynomial SUFFICIENT condition rather than the rank characterization. The note could not be obtained (IEEE paywall, 2026-09-11) so the second point is a lead, not a reading. The earlier A3 verdict remains correct about the AFP hits it inspected; its sentence 'nothing existing covers it' reached further than its search, which covered six corpora none of which is the wider Lean ecosystem. Evidence: docs/provenance/by001-by002-linear-systems-triage.md. READ 2026-09-11: the maintainer supplied the paper and both pages were read. Graded in section 25 of docs/provenance/source-coverage-audit.md - 0 Yes, 1 Partial, 16 No, 4 Beyond. The verdict is unchanged and is now checked rather than inferred: print states the property the ported criteria characterize, and none of the six statements it proves about that property is in the atlas. That sentence's successor is recorded above: those statements are now in the tree.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "isabelle-afp" ], "query_terms": [ "observability", "unobservable state", "state reconstruction", "observable output" ] }, "candidate_formalizations": [], "tags": [ "control-theory" ], "related_result_ids": [ "LAND-LINSYS-001" ], "public": { "group": "Limits of control and regulation", "title": "Unobservability", "summary": "Watching a system's outputs does not always tell you what state it was in, and there is an exact test for when it does.", "use": "Before claiming a system is monitorable: if the test fails, no amount of watching the outputs recovers the state.", "attribution": "Kalman; the equivalence with the dynamical property after Chen and Desoer, cited by Klamka 1972" } }, { "id": "BY-002", "name": "Uncontrollability of dynamical systems", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Some dynamical systems cannot be driven between arbitrary states by available controls.", "original_source_refs": [ "survey-ref-021", "survey-ref-024" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "module": "AISafetyAtlas.LinearSystems.Flow", "build_command": "lake build AISafetyAtlas.LinearSystems.Flow AISafetyAtlas.Examples.LinearSystems.Flow", "declaration": "AISafetyAtlas.LinearSystems.isCompletelyReachableSol_iff_isControllable", "relationship": "EXACT", "content_source_refs": [ "survey-ref-021" ], "scope_delta": { "summary": "The survey row asserts that some dynamical systems cannot be driven between arbitrary states by the available controls. IsCompletelyReachable is that property at the source's own quantifier - a run between ANY two states, not merely out of the origin - and isCompletelyReachable_iff_isControllable proves it equivalent to the Kalman controllability condition, with deaf_not_isCompletelyReachable a named system where it fails. THE CLASS IS THE SOURCE'S OWN, AND IS STATED HERE BECAUSE THE ROW'S SENTENCE DOES NOT CARRY IT: what is proved is the finite-dimensional linear time-invariant system x' = Ax + Bu over the complex numbers, the object of survey-ref-021 (Klamka 1972), and not dynamical systems at large. THE GRAMIAN IS NOT USED: reachedSet makes the states reached at a fixed time a subspace, an annihilating covector comes from the DUAL rather than an inner product, and the input that exposes the silence is the adjoint signal's own conjugate, so the integrand is a non-negative real. The one lemma Mathlib does not state in one piece is eqOn_zero_of_intervalIntegral_eq_zero. AT THE SOURCE'S SOLUTION CLASS (2026-10-05): the graded declaration is isCompletelyReachableSol_iff_isControllable, over IsSolution, the integral (Caratheodory) form that admits inputs with jumps, the piecewise-continuous class print imports from Chen and Desoer (a step input is witnessed, integrator_step_isSolution; that the whole class gives solutions of this kind is standard and not proved here); so the necessity direction covers bang-bang and piecewise-constant inputs, and the sufficiency direction produces a CONTINUOUS input. Reaching is forward in time (0 < t1). The classical-solution form isCompletelyReachable_iff_isControllable, on which the closure audit found the necessity side narrower than print, is kept beside it. pair_not_isCompletelyReachableSol is a non-degenerate witness (a real input that misses a state). NO TIME-WINDOW PARAMETER on this side, unlike the observability half, and section 25's Intro row discloses the asymmetry. NC-013 records that no proof assistant held this equivalence when the six-corpus sweep was run on 2026-09-20.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.LinearSystems.isCompletelyReachableSol_iff_isControllable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.LinearSystems.IsCompletelyReachableSol", "AISafetyAtlas.LinearSystems.IsSolution", "AISafetyAtlas.LinearSystems.IsControllable" ], "application": "Control: a linear time-invariant system can be driven from any state to any state, forward in time, EXACTLY WHEN the Kalman controllability condition holds, at the source's solution class (integral-form solutions, inputs with jumps admitted). The survey row's claim is the failing direction." }, { "atlas_declaration": "AISafetyAtlas.Examples.LinearSystems.pair_not_isCompletelyReachableSol", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Examples.LinearSystems.pairA", "AISafetyAtlas.Examples.LinearSystems.firstB" ], "application": "Control: a named system with a real input that still cannot reach every state, whatever locally integrable input is allowed." }, { "atlas_declaration": "AISafetyAtlas.LinearSystems.isCompletelyReachable_iff_isControllable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.LinearSystems.IsCompletelyReachable", "AISafetyAtlas.LinearSystems.drivenState", "AISafetyAtlas.LinearSystems.reachedSet", "AISafetyAtlas.LinearSystems.IsControllable" ], "application": "Control: a linear time-invariant system can be driven between ANY two states EXACTLY WHEN the Kalman controllability condition holds. The survey row's claim is the failing direction of this equivalence." }, { "atlas_declaration": "AISafetyAtlas.LinearSystems.not_isReachable_of_not_isControllable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.LinearSystems.IsReachable", "AISafetyAtlas.LinearSystems.eigenTrajectory" ], "application": "Control: when the Kalman condition fails, some state is not reachable from rest by any input. The survey row's sentence as a theorem." }, { "atlas_declaration": "AISafetyAtlas.Examples.LinearSystems.deaf_not_isCompletelyReachable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Examples.LinearSystems.integratorA", "AISafetyAtlas.Examples.LinearSystems.deafB" ], "application": "Control: a named system that cannot be driven between arbitrary states. The row's claim at a witness rather than in prose." } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison. RETIRED STATABILITY RECORD (2026-09-21). This row carried a statability verdict until it carried Lean; policy rejects a verdict on a row that has Lean, so the record is kept here rather than dropped. Its verdict was TRIAGED_DISTINCT and it remains correct about what it was about - the EXTERNAL candidate, AnandGokhale/LeanForControl, whose adapted layer is algebraic and is still not coverage of this row. What covers the row is atlas-original and was built on 2026-09-20. The note read: RETRIAGED 2026-09-13, from CANDIDATE_LEAD. Both of the two reasons the 2026-09-11 note gave for stopping at a lead are now discharged. THE SOURCE IS READ: Klamka's note was supplied by the maintainer and both printed pages were read on 2026-09-11 from rendered images, and section 25 of docs/provenance/source-coverage-audit.md grades it statement by statement. THE CANDIDATE IS ADJUDICATED AND FOUND DISTINCT: AnandGokhale/LeanForControl's LinearSystems track is REPRODUCED and in-tree as AISafetyAtlas.LinearSystems under LAND-LINSYS-001, and what it carries is the Kalman rank criteria and the Hautus tests - neither of which is any statement in print. WHAT WAS OWED AND IS NOW BUILT (2026-09-13): the previous note said the row is owed print's numbered results. They are in the tree, at print's own antecedent, in AISafetyAtlas.LinearSystems.BlockBound - not_isControllable_of_ceil_div_gt_rank and ..._gt_width, not_isObservable_of_ceil_div_gt_rank and ..._gt_height - reached through two lemmas Mathlib does not have, finrank_ker_comp_le and finrank_ker_pow_le, which are print's counting step with no Jordan form in it. THE TRAJECTORY BRIDGE IS BUILT AND THAT SENTENCE IS RETRACTED. The bridge landed 2026-09-20 and this note was swept 2026-09-21, one day late; the same one-day lag is recorded in section 25 of the coverage audit, which had already spent a week asserting a blocker it had itself removed. Until the sweep this note read: 'WHAT IS STILL NOT COVERED, AND IS WHY A FRESH FORMALIZATION REMAINS OWED: none of this defines a trajectory, a solution or an output signal ... and it is a substrate gap, not a candidate gap.' Every clause of it is now false. AISafetyAtlas.LinearSystems.Dynamics defines IsTrajectoryOn - a solution of x' = Ax + Bu asked on a set of times - and outputSignal, which is y = Cx; AISafetyAtlas.LinearSystems.Flow builds the matrix exponential, variation of constants and the adjoint argument. Print's two properties are in the tree AS PROPERTIES and each is EQUIVALENT to the algebraic criterion named for it, which is the equivalence page 726 states and attributes to Chen and Desoer. THE CONTROLLABILITY SIDE, WHICH IS THIS ROW. IsCompletelyReachable is print's own quantifier - a run between ANY two states, not only out of the origin - and isCompletelyReachable_iff_isControllable proves it equivalent to the Kalman rank condition on the controllability matrix. IsReachable and not_isReachable_of_not_isControllable are the reachable-from-rest form. AISafetyAtlas.Examples.LinearSystems.deaf_not_isCompletelyReachable exhibits a system that cannot be driven between arbitrary states, which is this row's informal claim at a witness. NC-013 in docs/provenance/formalization-search.json records that no proof assistant held the continuous-time reachability equivalence when the six-corpus search was run that morning. The observability side is on BY-001. PROMOTED 2026-09-21, on the maintainer's decision. This note recorded the promotion as open, because moving a survey row to covered moves the headline coverage counts. The row now carries lean_artifact and a formalization record graded EXACT. The class restriction the row's own sentence does not carry is stated in the scope_delta rather than left implicit: what is proved is the finite-dimensional linear time-invariant system over the complex numbers, which is the object of survey-ref-021 and not dynamical systems at large. THE IDENTIFICATION THAT WAS ASSUMED IS NOW PROVED (2026-09-20): print's index is the multiplicity in the minimal polynomial and the exponent in the new statements was characterized by stabilization of the generalized eigenspace chain. The two are equal; print cites Zadeh and Desoer for it and does not prove it, and maxGenEigenspaceIndex_eq_rootMultiplicity_minpoly in AISafetyAtlas.LinearSystems.BlockBound now does, over an arbitrary field and with no Jordan form. Print's numbered results are available with print's own nu computed from A. Print's Theorem 1 and Corollaries 1, 3 and 4 are the controllability side and are the ones this row is stated against. PREVIOUS NOTE, KEPT: Triaged 2026-09-11, the first triage this row has had. AnandGokhale/LeanForControl formalizes the Kalman rank criterion and the Hautus test for controllability; its LinearSystems track is now in-tree as AISafetyAtlas.LinearSystems under LAND-LINSYS-001. The recorded corpus sweep reported zero candidates because none of its six corpora is the wider Lean ecosystem, not because none exists. CANDIDATE_LEAD and not coverage: the ported layer defines no trajectory, and Klamka's 1972 note has not been read. Evidence: docs/provenance/by001-by002-linear-systems-triage.md. READ 2026-09-11: the maintainer supplied the paper and both pages were read. Graded in section 25 of docs/provenance/source-coverage-audit.md - 0 Yes, 1 Partial, 16 No, 4 Beyond. The verdict is unchanged and is now checked rather than inferred. That sentence's successor is recorded above: those statements are now in the tree.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "uncontrollability", "not controllable", "controllability" ] }, "tags": [ "control-theory" ], "related_result_ids": [ "LAND-LINSYS-001" ], "public": { "group": "Limits of control and regulation", "title": "Uncontrollability", "summary": "Some systems cannot be driven from where they are to where you want them, whatever inputs you apply, and there is an exact test for which ones.", "use": "Before claiming a system is steerable: if the test fails, no control law reaches the states it excludes.", "attribution": "Kalman; the equivalence with the dynamical property after Chen and Desoer, cited by Klamka 1972" } }, { "id": "BY-003", "name": "Good Regulator Theorem", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Every maximally simple optimal regulator must embody a model of the regulated system.", "original_source_refs": [ "survey-ref-025" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "good regulator theorem", "good regulator" ] }, "tags": [ "control-theory" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "RETRIAGED 2026-09-13, from UNTRIAGED. THE SOURCE IS NOW HELD AND READ. Conant and Ashby was fetched 2026-09-13 from the Principia Cybernetica deposit at VUB and is pinned as conant-ashby-ijss-1970-every-good-regulator-of-a-system-must-be-a-model-of-that-system.pdf, sha256 fba0430f1196748e81915a11860948b5dbb249e861b9a086953b67420362ec7d, manifested 2026-09-13. The face reads 'Int. J. Systems Sci., 1970, vol. 1, No. 2, 89-97', which matches survey-ref-025. Journal page 96 read from a rendered page image. WHAT PRINT ACTUALLY STATES, and it is NOT numbered: 'Theorem: The simplest optimal regulator R of a reguland S produces events R which are related to the events S by a mapping h : S -> R.' Optimal means the regulator's conditional distribution p(R|S) minimises H(Z), where Psi : R x S -> Z is given and p(S) is fixed; simplest means least H(R). The heart of the proof is a lemma print states explicitly: for every s_j in S the set of psi(r_i, s_j) over r_i with p(r_i, s_j) > 0 has exactly ONE element. SO THE OBJECT IS ENTROPY, NOT CYBERNETIC PROSE, and the lemma is the formalizable core: at an entropy-minimising conditional distribution the regulator is effectively deterministic given the reguland. Its proof is strict concavity of entropy under a mass transfer between two outcomes. NO CANDIDATE ANYWHERE: the six-corpus sweep on this row returned zero, and AISafetyAtlas/Control.lean says in its own docstring that the Good Regulator theorem carries no Lean on any branch - checked 2026-09-13 and true. What the tree does have next door is Ashby's law of requisite variety, ashby_logVariety_ge and ashby_logVariety_ge_mul in AISafetyAtlas.Control.VarietyCounting, and channel capacity in AISafetyAtlas.InformationTheory; those are a DIFFERENT theorem of the same authorship and are not this one. SO A FRESH FORMALIZATION IS OWED and the substrate exists: a finite joint distribution, Shannon entropy and its strict concavity, all reachable from the pinned Mathlib and the vendored PFR. What is NOT settled is whether print's 'simplest' - minimal H(R) among the optimal class - is needed for the lemma or only for the theorem; print's proof structure suggests the lemma needs only optimality, and that is a reading to check against the rest of journal pages 96-97, which were not read." } }, { "id": "BY-004", "name": "Law of Requisite Variety", "paper_reference": "Table 1", "survey_proof_assessment": "CONDITIONAL", "informal_claim": "Perfect regulation requires sufficient controller variety under perfect information and unbounded response speed.", "original_source_refs": [ "survey-ref-026" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.ChannelRate", "declaration": "AISafetyAtlas.Control.chainRate", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-026" ], "content_source_note": "Section 9/12's definition of the entropy of one step of a Markov chain: the average of the columns' entropies, each weighted by the proportion in which that state occurs when the sequence has settled to its equilibrium. Written from an equilibrium law and a transition kernel, with no sample space.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.ChannelRate AISafetyAtlas.Examples.Control.ChannelRate", "scope_delta": { "summary": "The quantity section 9/15 calls channel capacity, which channelCapacity (the noiseless alphabet ceiling log |O|) is not. chainRate_eq_condEntropy identifies it with H[X1 | X0] on the canonical one-step chain. Checked against Ashby's own worked three-state insect chain, whose equilibrium is exactly (22/49, 21/49, 6/49) against his printed 0.449, 0.429, 0.122.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.ChannelRate", "declaration": "AISafetyAtlas.Control.entropy_traj", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-026" ], "content_source_note": "Section 9/15: 'the entropy of a length of Markov chain is proportional to its length (provided always that it has settled down to equilibrium)'.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.ChannelRate AISafetyAtlas.Examples.Control.ChannelRate", "scope_delta": { "summary": "Wider than print in two ways. The source asserts this without proof; it is proved here for a Markov stationary process. And the printed word 'proportional' is loose: the exact identity is H[X_0 ... X_n] = H[X_0] + n * rate, affine rather than linear. The two agree when H[X_0] equals the rate, which is what happens in the spun-coin case Ashby checks it against, so his own example cannot separate them. Same class of printed slip as section 11/8's H_D(E).", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.ChannelRate", "declaration": "AISafetyAtlas.Control.entropy_outcome_ge_sub_chainEntropy", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-026" ], "content_source_note": "Section 11/11: 'R's capacity as a regulator cannot exceed R's capacity as a channel of communication', with capacity taken to be section 9/15's entropy rate rather than the alphabet ceiling used by the section's exercises.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.ChannelRate AISafetyAtlas.Examples.Control.ChannelRate", "scope_delta": { "summary": "Bounds the outcome entropy by Ashby's own entropy-rate capacity rather than by the alphabet ceiling, which is what section 11/11 asserts. The budget is the exact trajectory entropy H[X_0] + n * rate, not rate times time; a bare capacity-times-time budget is not an upper bound, because the initial state carries entropy of its own. entropy_outcome_ge_sub_channelCapacity is retained as the sharper form for the four noiseless exercises.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.CompleteControl", "declaration": "AISafetyAtlas.Control.outcome_eq_comp", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-026" ], "content_source_note": "Section 11/14: 'the fact that R is a perfect regulator gives C complete control over the output, in spite of the entrance of disturbing effects by way of D. Thus, perfect regulation of the outcome by R makes possible a complete control over the outcome by C.'", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.CompleteControl AISafetyAtlas.Examples.Control.CompleteControl", "scope_delta": { "summary": "Same as print. Under perfect regulation the outcome is a function of the controller's setting alone: the disturbance is eliminated rather than bounded. exists_strategy_forcing is complete control, and seq_outcome_eq is the printed compound target a, b, a, c, c, a produced regardless of D's values during the sequence.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.CompleteControl", "declaration": "AISafetyAtlas.Control.card_disturbance_le_card_regulator", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-026" ], "content_source_note": "Section 11/14: 'The achievement of control may thus depend necessarily on the achievement of regulation.'", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.CompleteControl AISafetyAtlas.Examples.Control.CompleteControl", "scope_delta": { "summary": "Wider than print. The printed sentence is qualitative; here it is a count, and a count derived from section 11/10's own impossibility theorem rather than from a fresh argument. Perfect regulation collapses admittedOutcomes to a singleton, which two_le_card_admittedOutcomes forbids below a threshold of regulator variety, so |D| <= |R| and, by card_controller_le_card_regulator, |C| <= |R|. Requisite variety is charged twice, once for each of the regulator's two inputs.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.CompleteControl", "declaration": "AISafetyAtlas.Control.max_channelCapacity_le_channelCapacity_regulator", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-026" ], "content_source_note": "Section 11/14, Ex. 4: 'R to T must carry 2 bits/sec to neutralise D (from Ex. 2), and 20 bits/sec from C; as these two are independent (D's values and C's not correlated), the capacity must be at least 22 bits/sec.' Read from the Martino Fine Books 2015 reprint, book p. 285; the pinned scan omits Answers to Exercises, pp. 274 to 288.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.CompleteControl AISafetyAtlas.Examples.Control.CompleteControl", "scope_delta": { "summary": "Narrower than the printed answer, and deliberately. What the model forces is the maximum of the two loads, not their sum. Additivity is a property of the table rather than of the diagram: Examples.Control.ashbyControl_capacity_lt_sum runs on Ashby's own answer to Ex. 1, a perfect regulator on Table 11/3/1 whose entire repertoire is log 3 where the additive reading demands log 9. This is not a correction to his arithmetic, since Ex. 4 inherits Ex. 2's attenuating T while Table 11/3/1 attenuates nothing; it is a limit on what section 11/14's diagram alone implies. channelCapacity_prod is the lemma that licenses the sum where the R to T link really is two independent sub-channels. Capacities here are per use, whereas Ashby quotes bits per second; ashbyCapacity converts.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.CompleteControl", "declaration": "AISafetyAtlas.Control.entropy_outcome_eq_entropy_controller", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-026" ], "content_source_note": "Section 11/14: a suitable regulator R 'may be able to form, with T, a compound channel to E that transmits fully from C while transmitting nothing from D.'", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.CompleteControl AISafetyAtlas.Examples.Control.CompleteControl", "scope_delta": { "summary": "Both clauses proved. This one is full transmission from C, when distinct targets name distinct outcomes. condEntropy_outcome_controller is transmission of nothing from D and needs no hypothesis on the disturbance's law at all; mutualInfo_outcome_disturbance_eq_zero is the form under Ashby's own side condition that D's values and C's are not correlated. The two-input factorization FactorsThrough, used by the Ex. 2 and Ex. 3 capacity bounds, is the atlas's reading of the printed diagram and is a hypothesis, not a consequence.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.RequisiteVariety", "declaration": "AISafetyAtlas.Control.ashby_variety_ge", "relationship": "EXACT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.RequisiteVariety AISafetyAtlas.Examples.Control.RequisiteVariety" }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.RequisiteVariety", "declaration": "AISafetyAtlas.Control.ashby_logVariety_ge", "relationship": "EXACT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.RequisiteVariety AISafetyAtlas.Examples.Control.RequisiteVariety" }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.RequisiteVariety", "declaration": "AISafetyAtlas.Control.card_le_mul_card_admittedOutcomes_mul", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.RequisiteVariety AISafetyAtlas.Examples.Control.RequisiteVariety", "scope_delta": { "summary": "Unifies chapter 11's sections 11/5 and 11/9 into one statement with the repetition count k as a parameter, where the source argues them separately; and weakens the column condition from injectivity of every column to Set.InjOn on the fibre a strategy actually visits, so the theorem covers tables the printed statement excludes. Section 11/9's printed bound writes log V_R for an already-logarithmic V_R; the multiplicative form stated here is the reading its derivation supports.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.RequisiteVariety", "declaration": "AISafetyAtlas.Control.two_le_card_admittedOutcomes", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.RequisiteVariety AISafetyAtlas.Examples.Control.RequisiteVariety", "scope_delta": { "summary": "The impossibility reading of section 11/10 ('the law states that certain events are impossible'), which the source states in prose rather than as a proposition: a regulator with strictly fewer moves than there are disturbances cannot hold the outcome fixed under any strategy. Derived from the counting law; not a separate printed statement.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.RequisiteVariety", "declaration": "AISafetyAtlas.Control.entropy_ge_of_condEntropy_ge", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.RequisiteVariety AISafetyAtlas.Examples.Control.RequisiteVariety", "scope_delta": { "summary": "Section 11/8 with the 11/9 slack K, over any measurable space with a zero-or-probability measure rather than Shannon's discrete sources. Not graded EXACT because the printed conclusion is misprinted: it reads H_D(E) where the derivation above it and the sentence below it both require H_D(R). The corrected inequality is what is proved. The 1961 typesetting also renders the relation as strict throughout; the non-strict reading is the only one the derivation supports.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.RequisiteVariety", "declaration": "AISafetyAtlas.Control.entropy_outcome_ge_of_strategy", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.RequisiteVariety AISafetyAtlas.Examples.Control.RequisiteVariety", "scope_delta": { "summary": "Section 11/8's headline H(E) >= H(D) - H(R) for a determinate strategy. Stronger than the printed statement in that the source assumes H_R(E) >= H_R(D) as a hypothesis, whereas condEntropy_outcome_eq derives it — as an equality — from the table's column condition, tying the combinatorial and entropy halves of the chapter into one development.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.RequisiteVariety", "declaration": "AISafetyAtlas.Control.entropy_ge_of_sensor", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.RequisiteVariety AISafetyAtlas.Examples.Control.RequisiteVariety", "scope_delta": { "summary": "This row's informal claim 'under perfect information' matches the counting sections 11/5-11/10 rather than 11/11. The claim's other qualifier, unbounded response speed, has no counterpart — chapter 11 is untimed and nothing here speaks to latency. This is the deterministic sensor-reading consequence: the bound uses the entropy of one observation, not Ashby's general channel-capacity rate. channelCapacity later added the noiseless alphabet ceiling used by the exercises, but it does not close the general section 11/11 claim.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.RequisiteVariety", "declaration": "AISafetyAtlas.Control.entropy_outcome_ge_sub_channelCapacity", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.RequisiteVariety AISafetyAtlas.Examples.Control.RequisiteVariety", "scope_delta": { "summary": "A noiseless alphabet-ceiling consequence of section 11/11. channelCapacity O = log |O| and the rate lemmas reproduce the section's signal-counting exercises. This declaration on its own does not reproduce Ashby's general capacity: section 9/15 first computes a three-state Markov source at 0.842 bits per step and 2.53 bits per minute, then calls such a rate the natural measure of channel capacity; log |O| would instead give log 3 per step. That quantity is now declared rather than proxied — chainRate is section 9/12's equilibrium-weighted column entropy, chainRate_eq_condEntropy identifies it with the conditional entropy of one step, and entropy_outcome_ge_sub_chainEntropy carries the general claim — so the two capacities are formalized side by side, the alphabet ceiling here for the section's signal-counting exercises and the rate there for the general statement. The source-audit row is therefore Yes | Wider. Shannon's noisy capacity sup I(input : output) is still not modelled, and section 11/11 does not need it. The observation-variable model is wider than a deterministic sensor. The statement is necessary only; nothing here shows a channel of sufficient capacity admits a regulator that achieves constancy.", "evidence": "docs/provenance/ashby-requisite-variety.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.RequisiteVariety", "declaration": "AISafetyAtlas.Control.entropy_le_channelCapacity_of_complete", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.RequisiteVariety AISafetyAtlas.Examples.Control.RequisiteVariety", "scope_delta": { "summary": "The form section 11/11's exercises use: complete regulation - the outcome held at a single value, Ashby's 'it is constancy that is to be transmitted' - forces the disturbance entropy below the channel capacity. Necessary only, and the three worked exercises are graded accordingly. Examples.Control.ship_disturbance_upper_limit answers Ex. 2, whose printed question - given that the telegraph and wheel are normally sufficient for full regulation, estimate an upper limit for the disturbances - is exactly what the law answers: log 9 + 5 log 50 in five seconds, counted by adding the two controls' capacities. Examples.Control.insect_optic_nerve_not_binding answers Ex. 1 by showing the constraint does not bind (2000 log 2 a second against ten dangers at 10 log 2), which is NOT an answer to the printed question 'is this sufficient to enable it to defend itself'. Examples.Control.general_intelligence_insufficient answers Ex. 3 in the direction the law does settle: ten signallers carrying 576000 log 2 a day cannot regulate ten divisions manoeuvring at 10^7 log 2, so complete regulation is impossible.", "evidence": "docs/provenance/ashby-requisite-variety.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Control.ashby_variety_ge", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.admittedOutcomes", "AISafetyAtlas.Control.card_le_mul_card_admittedOutcomes" ] }, { "atlas_declaration": "AISafetyAtlas.Control.card_ceilDiv_le_admittedOutcomes", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.admittedOutcomes", "AISafetyAtlas.Control.card_le_mul_card_admittedOutcomes" ] }, { "atlas_declaration": "AISafetyAtlas.Control.ashby_variety_ge_isSharp", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.shiftTable", "AISafetyAtlas.Control.card_admittedOutcomes_shiftTable" ] }, { "atlas_declaration": "AISafetyAtlas.Control.ashby_logVariety_ge_mul", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.card_le_mul_card_admittedOutcomes_mul" ] }, { "atlas_declaration": "AISafetyAtlas.Control.ashby_logVariety_ge", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.admittedOutcomes", "AISafetyAtlas.Control.card_le_mul_card_admittedOutcomes" ] }, { "atlas_declaration": "AISafetyAtlas.Control.card_le_mul_card_admittedOutcomes_mul", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.admittedOutcomes" ] }, { "atlas_declaration": "AISafetyAtlas.Control.two_le_card_admittedOutcomes", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.card_le_mul_card_admittedOutcomes" ] }, { "atlas_declaration": "AISafetyAtlas.Control.entropy_ge_of_condEntropy_ge", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Control.condEntropy_outcome_eq", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Control.entropy_outcome_ge", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropy_ge_of_condEntropy_ge", "AISafetyAtlas.Control.condEntropy_outcome_eq" ] }, { "atlas_declaration": "AISafetyAtlas.Control.entropy_outcome_ge_of_strategy", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropy_outcome_ge", "AISafetyAtlas.InformationTheory.condEntropy_comp_self_left" ] }, { "atlas_declaration": "AISafetyAtlas.Control.entropy_ge_of_sensor", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropy_outcome_ge_of_strategy" ] }, { "atlas_declaration": "AISafetyAtlas.Control.entropy_outcome_ge_sub_channelCapacity", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropy_outcome_ge", "AISafetyAtlas.InformationTheory.channelCapacity" ], "application": "Regulation under limited information: in the noiseless alphabet-ceiling case used by Ashby's section 11/11 exercises, a regulator cannot reduce the outcome entropy below the disturbance entropy by more than log |O| per channel use." }, { "atlas_declaration": "AISafetyAtlas.Control.entropy_le_channelCapacity_of_complete", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropy_outcome_ge_sub_channelCapacity" ] }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.channelCapacity_fun", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.channelCapacity_eq_of_card_eq_pow" ] }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.channelCapacity_prod", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.channelCapacity" ] }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.channelCapacity_eq_of_card_eq_two_pow", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.channelCapacity_eq_of_card_eq_pow" ] }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.channelCapacity_eq_of_card_eq_pow", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.channelCapacity" ] }, { "atlas_declaration": "AISafetyAtlas.Oversight.not_forces_of_card_lt", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Control.two_le_card_admittedOutcomes" ], "application": "The AI-system reading of the law: an oversight regime whose interventions are fewer than the situations it is answerable for cannot hold the outcome to a single target, whatever it observes. Necessary condition only, and only for forcing every situation with certainty. Reviewed 2026-08-17; the signature is scoped to this bridge and not to Ashby's law in general. Package: docs/interpretation-reviews/review-oversight-varietybound.md.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-08-17", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-oversight-varietybound.md" } } ] }, "root_import": true, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "REVIEWED", "interpretation_review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-08-17", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-oversight-varietybound.md" }, "notes": "Source comparison done against W. Ross Ashby, An Introduction to Cybernetics, including sections 9/15 and 11/5 to 11/11; recorded in docs/provenance/ashby-requisite-variety.md, which also lists three clashes in the printed text (a misprinted conclusion in 11/8, strict relation symbols throughout, and a double logarithm in 11/9). The survey's citation transposes the author's initials as 'R. W. Ashby'; survey-ref-026 records the survey's text verbatim and is not corrected. Section 11/11 is covered in both readings. Section 11/14 is covered by AISafetyAtlas.Control.CompleteControl: IsPerfectRegulator is its hypothesis, outcome_eq_comp its first half, and card_disturbance_le_card_regulator with card_controller_le_card_regulator turn its closing sentence about control depending on regulation into a count, derived from section 11/10's impossibility theorem. The noiseless alphabet/exercise reading is channelCapacity O = log |O|, which is what the four printed exercises compute. Section 9/15's general reading is an entropy rate: its worked three-state Markov example uses the probability-weighted entropy 0.842 bits per step rather than log 3, before naming such a rate the natural measure of capacity. AISafetyAtlas.Control.ChannelRate supplies that: chainRate is section 9/12's weighted column entropy, chainRate_eq_condEntropy identifies it with H[X1|X0], ashbyCapacity is the per-unit-time rate, and entropy_outcome_ge_sub_chainEntropy is section 11/11's bound against it. entropy_traj proves section 9/15's asserted proportionality and sharpens it from linear to affine. The definition is checked against Ashby's own chain, whose equilibrium is exactly (22/49, 21/49, 6/49) against his printed 0.449, 0.429, 0.122. The row is therefore Yes: the atlas bounds the regulator by the capacity Ashby names, and the definition is pinned to his arithmetic rather than to a reconstruction of it. Shannon's noisy capacity sup I(input : output) is still not modelled and section 9/15 does not ask for it. entropy_outcome_ge_sub_channelCapacity is a necessary alphabet-ceiling bound and entropy_le_channelCapacity_of_complete is the form the exercises use; sufficiency is neither proved nor implied. Maintainer bridge review 2026-08-17 (REVIEWED): applies to the AI-facing bridge Oversight.not_forces_of_card_lt, which this row owns - statement and scoped interpretation accepted per docs/interpretation-reviews/review-oversight-varietybound.md. It is NOT a reviewed AI-system reading of Ashby's law in general, and it licenses none of that package's forbidden claims: not that oversight does not work, not that monitoring is useless, not anything about a deployed system, not that more interventions suffice. The accepted reading is a conditional whose antecedent - fewer interventions distinct in effect than situations - nobody has established for any real overseer.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "requisite variety", "law of requisite variety" ] }, "tags": [ "control-theory", "information-theory" ], "public": { "group": "Limits of control and regulation", "summary": "A controller cannot hold things steadier than its own range of responses allows.", "use": "Sizing what a safety controller has to be able to do.", "title": "Requisite variety", "attribution": "Ashby" }, "escape_routes": [ { "axis": "ADD_RESOURCE", "status": "NAMED_ONLY", "note": "Ashby's bound is an inequality in regulator variety and channel capacity, so more of either is the direction out. This is deliberately NOT graded FORMALIZED: `entropy_le_channelCapacity_of_complete` states its own limit -- it is a necessary condition only, and its docstring says that a channel of sufficient capacity is not thereby shown to admit a regulator achieving constancy, because nothing here constructs one. The route is the contrapositive of a bound, not a construction." } ] }, { "id": "BY-005", "name": "Information-theoretical control limits", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Information constraints impose lower bounds on achievable control.", "original_source_refs": [ "survey-ref-027" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.InformationLimits", "declaration": "AISafetyAtlas.Control.entropy_noise_sub_controlLoss", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 2 of arXiv:physics/0104007v2, strengthened to an exact identity: H(Z) - L_C = H(Z | X', X, C). The printed inequality is that quantity being nonnegative and the printed equality condition is it being zero, so both are corollaries (controlLoss_le_entropy_noise, controlLoss_eq_entropy_noise_iff) rather than separate arguments.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.InformationLimits AISafetyAtlas.Examples.Control.InformationLimits", "scope_delta": { "summary": "The published Physica A text has now been read and its Theorem 2 matches the preprint's exactly. RELATED, not EXACT, for a substantive reason: the source defines L_C with a minimization over all conditional distributions {p(c|x)}, 'to ensure that L_C reflects the properties of the actuation channel and does not depend on one's choice of control inputs'. The atlas controlLoss is the unminimized H(X'|X,C), so every statement here is the pointwise-in-controller version, which implies the printed one at the minimizer but is not the same statement. This declaration is additionally a strengthening: the printed Theorem 2 is an inequality plus a separate equality condition, and both are corollaries of the identity proved here. Also holds over an arbitrary measurable space rather than the source's finite alphabets. Updated 2026-08-16: minControlLoss now defines eq. (28) itself and minControlLoss_le_entropy_noise proves Theorem 2's INEQUALITY of it, so the minimization caveat no longer applies to that half. It still applies to the equality case, which describes a minimizer that sInf need not have; that half stays pointwise as controlLoss_eq_entropy_noise_iff.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.InformationLimits", "declaration": "AISafetyAtlas.Control.controlLoss_eq_condMutualInfo", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 3: L_C = I(X' : Z | X, C). Uses the mutual-information chain rule added in AISafetyAtlas.InformationTheory.DataProcessing, which PFR does not carry.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.InformationLimits AISafetyAtlas.Examples.Control.InformationLimits", "scope_delta": { "summary": "Theorem 3 of the published text, verbatim in content. The source's L_C carries a minimization over {p(c|x)}; this is the pointwise-in-controller statement, which is the STRONGER of the two, since it holds at every controller rather than at one. The printed statement is minControlLoss_eq_sInf_condMutualInfo, derived from this one by taking infima. RELATED rather than EXACT because the model of a policy set is the atlas's, not the paper's, and because the paper's controllability development is not mechanized — Theorem 1 characterizes perfect controllability and is deliberately omitted. The observability half of that earlier reason is stale and is withdrawn: AISafetyAtlas.Control.Observability mechanizes Theorem 5 as perfectlyObservable_iff_sensorLoss_eq_zero, Theorem 6 as condMutualInfo_eq_zero_of_sensorLoss_eq_zero, and Corollary 7 as mutualInfo_prod_eq_of_sensorLoss_eq_zero. Generalized from finite alphabets to any measurable space with a zero-or-probability measure.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.InformationLimits", "declaration": "AISafetyAtlas.Control.controlLoss_eq_mutualInfo_sub", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 4: L_C = I(X' : X,C,Z) - I(X' : X,C).", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.InformationLimits AISafetyAtlas.Examples.Control.InformationLimits", "scope_delta": { "summary": "Theorem 4 of the published text, verbatim in content. As for the Theorem 3 record: this is the pointwise-in-controller form, which is the stronger one, and the printed minimized statement is minControlLoss_eq_sInf_mutualInfo_sub. Generalized from finite alphabets to any measurable space with a zero-or-probability measure.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.InformationLimits", "declaration": "AISafetyAtlas.Control.entropyReduction_le_of_condEntropy_ge", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 10, the paper's stated main result: closed-loop control beats the best open-loop control by at most the information the controller gathered, so one bit measured is worth at most one bit of control.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.InformationLimits AISafetyAtlas.Examples.Control.InformationLimits", "scope_delta": { "summary": "The paper's step (50) — that a closed-loop controller is formally equivalent to an ensemble of open-loop controllers acting on the conditional supports — is justified in the source in one sentence beneath the proof: each conditional distribution p(x|c) is a legitimate input distribution, an element of P. entropyReduction_le_of_condEntropy_ge takes the step as an explicit hypothesis and asserts nothing about the model; entropyReduction_le_of_openLoopBound derives it from IsPlant and OpenLoopBound, which are atlas renderings rather than formulas the paper writes. OpenLoopBound is not the printed Delta-H-open-max: conditioning can change the joint law of state and noise, while the printed maximum varies the input law with a transition kernel fixed. openLoopMax narrows that mismatch by ranging over all input laws inside an independent-noise realization X'=F(X,C,Z), and isPurification_purifyMap now realizes every printed transition kernel that way, with openLoopMax_purifyMap showing the two reduction sets coincide; isGreatest_kernelOpenLoopMax proves the supremum attained, so it is the printed maximum. Theorem 10 is therefore Yes | Wider, stated against the realized maximum by exists_kernelEntropyReduction_le_at_max, which does not go through OpenLoopBound. OpenLoopBound is kept because its conditional-ensemble hypothesis is incomparable with that rendering. Theorems 5, 6 and 9 and Corollary 7 and Lemma 8 are formalized; Theorems 1 and 11 remain unformalized, and the quantum sections are out of scope.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.OpenLoopAttainment", "declaration": "AISafetyAtlas.Control.isGreatest_kernelOpenLoopMax", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Equation (48), Delta H_open^max = max over p_X in P and c in C, with P stated below the equation as the set of all probability distributions for the initial state. The printed notation asserts the maximum is attained.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.OpenLoopAttainment AISafetyAtlas.Examples.Control.OpenLoopAttainment", "scope_delta": { "summary": "The atlas previously rendered equation (48) as sSup with boundedness, which is formally weaker than print because max <= sSup. This proves the supremum is an element of the set, so the rendering IS the printed maximum. Probability measures on a finite state space are exactly the points of Mathlib's standard simplex via weights and ofWeights; in those coordinates the reduction is a finite sum of negMulLog terms whose inner argument is linear in the weights, by comp_ofWeights_real; continuity plus simplex compactness plus finiteness of the action alphabet give attainment.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.OpenLoopAttainment", "declaration": "AISafetyAtlas.Control.exists_kernelEntropyReduction_le_at_max", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 10 with its right-hand side written against the attained maximum of equation (48) rather than against a supremum that bounds the family.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.OpenLoopAttainment AISafetyAtlas.Examples.Control.OpenLoopAttainment", "scope_delta": { "summary": "The form equation (48)'s max notation promises: an explicit input distribution and action realize the bound. Closes the last residual recorded against the Theorem 10 row.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.Purification", "declaration": "AISafetyAtlas.Control.isPurification_purifyMap", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Equation (7) and the two conditions stated above it in Section 2: (i) the extended transition matrix p(x'|x,c,z) is deterministic for all c and z, and (ii) tracing out z reproduces p(x'|x,c) = sum_z p(x'|x,c,z) p_Z(z). Condition (i) is carried by the type of the map rather than by a hypothesis. Condition (ii) is IsPurification.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.Purification AISafetyAtlas.Examples.Control.Purification", "scope_delta": { "summary": "The paper asserts this representation for any non-deterministic channel and never constructs it. This constructs it, reading the paper's own phrase 'a randomly selected deterministic channel' literally: the seed is a random element of the finite space S x K -> T, drawn columnwise from the kernel, so no continuum seed is needed. Stronger than print, which offers the remark without a witness.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.Purification", "declaration": "AISafetyAtlas.Control.kernelEntropyReduction_le_kernelOpenLoopMax", "relationship": "EXACT", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 10, Delta H_closed <= Delta H_open^max + I(X;C), stated from the printed data alone: a joint law rho for the state and the action, and a Markov kernel kappa playing p(x'|x,c). No plant map and no seed occur in the statement.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.Purification AISafetyAtlas.Examples.Control.Purification", "statement_match_note": "Graded EXACT on 2026-08-17 after the two residuals that had held it at RELATED were both closed. The family residual - that the earlier entropyReduction_le_openLoopMax held only inside the independent-noise realization X'=F(X,C,Z) - is closed by openLoopMax_purifyMap, which shows the two renderings of equation (48) generate the same set of reductions, so their suprema agree exactly. The sSup-versus-max residual is closed by isGreatest_kernelOpenLoopMax, which proves the supremum attained, so kernelOpenLoopMax is equation (48)'s printed maximum rather than a weaker supremum. No residual against the printed statement remains; the audit grades this printed row Yes and Wider." }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.Purification", "declaration": "AISafetyAtlas.Control.kernelEntropyReduction_le_iSup_kernelOpenLoop", "relationship": "EXACT", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 9, Delta H_open <= max over c of Delta H_open^c, for an open-loop controller whose action law p_C is independent of the state, stated for an arbitrary Markov kernel.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.Purification AISafetyAtlas.Examples.Control.Purification", "statement_match_note": "Graded EXACT on 2026-08-17. Both sides are stated from the printed data alone - an input law, an action law independent of the state, and an arbitrary Markov kernel - with no plant model in the statement. The one recorded obstruction was the claim that the source does not assume purification in the open-loop section; Section 2 of the published text states it for any non-deterministic channel, before Section 3, so that claim was false and is withdrawn. The printed attainment clause is separate and proved by exists_kernelEntropyReduction_dirac_eq_iSup. The audit grades this printed row Yes and Wider." }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.Purification", "declaration": "AISafetyAtlas.Control.exists_kernelEntropyReduction_dirac_eq_iSup", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 9's attainment clause: the subdynamics at c-hat = arg max over c of Delta H_open^c is open-loop optimal. A pure controller is a Dirac action law.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.Purification AISafetyAtlas.Examples.Control.Purification", "scope_delta": { "summary": "Attainment here is immediate because the supremum is over the finite action alphabet K; equation (48)'s supremum over all input distributions needs a compactness argument and is proved attained separately by isGreatest_kernelOpenLoopMax.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.PolicyKernel", "declaration": "AISafetyAtlas.Control.kernelMinControlLoss", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Equation (28) as the source writes it: L_C = min over the stochastic kernels {p(c|x)} of H(X'|X,C). Declared over Mathlib Kernel S K with IsMarkovKernel, realized as mu compProd kappa.comap X on Omega times K, so the action is drawn from the state and nothing else. This is the printed object rather than a rendering of it on a fixed sample space.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.PolicyKernel AISafetyAtlas.Examples.Control.PolicyKernel", "scope_delta": { "summary": "The source's minimization itself, not a represented family standing in for it. minControlLoss_inputPolicies_eq_kernelMin proves it equal to the realized infimum over inputPolicies under finite alphabets, which is the printed setting; that closes what the provenance record had listed as the first of two reasons every record here is RELATED. The proof does not construct a kernel realization. Equation (28)'s own second displayed line is linear in p(c|x) because the numbers H(X'|x,c) belong to the actuation channel, so the minimum sits at a vertex of the simplex; the vertices are deterministic state feedbacks, which need no auxiliary randomness and so exist on every sample space. kernelControlLoss_eq_sum states that displayed line verbatim. minControlLoss_inputPolicies_attained exhibits the minimizer, which the source itself never does - it writes min where its own argument supports inf.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.OpenLoop", "declaration": "AISafetyAtlas.Control.openLoopMax", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "A bounded-supremum rendering of equation (48): every probability measure on the state space and every control value are ranged over, with an independent noise law held fixed. The source instead starts from an arbitrary fixed transition kernel and writes a maximum.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.OpenLoop AISafetyAtlas.Examples.Control.OpenLoop", "scope_delta": { "summary": "A closer rendering than OpenLoopBound, but not the printed object at full scope. openLoopMax varies the input law while holding the independent-noise subdynamics F and eta fixed, so each conditional law p(x|c) is genuinely included. Boundedness is load-bearing because Real.sSup of an unbounded set is 0; bddAbove_openLoopReductions supplies it from Fintype S. Two gaps once stood here and both are closed. First, the source fixes an arbitrary transition kernel p(x'|x,c) while the atlas assumes a deterministic F driven by a common noise seed independent of state and action; isPurification_purifyMap bridges the classes and openLoopMax_purifyMap shows the two reduction sets coincide. The independence cited at source step (30) belongs to Theorem 2, not Theorem 10, but Section 2 states purification globally for any channel. Second, the source writes max while the atlas had only a bounded sSup; isGreatest_kernelOpenLoopMax proves attainment. The source-audit row is Yes | Wider. entropyReduction_le_of_openLoopBound is kept alongside it because its hypothesis is incomparable.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.InformationLimits", "declaration": "AISafetyAtlas.Control.minControlLoss", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Equation (28) itself, the definition of the average control loss: L_C = min over {p(c|x)} of H(X' | X, C). Not a theorem and so uses no chain rule. The atlas feasible set is a parameter; inputPolicies is the measurable conditional-independence rendering on the current sample space, and NOT Set.univ. Its infimum is proved equal to the all-kernel minimum for finite alphabets by minControlLoss_inputPolicies_eq_kernelMin, against the object kernelMinControlLoss; that equality is not part of this declaration, which is why the parameterization is kept.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.InformationLimits AISafetyAtlas.Examples.Control.InformationLimits", "scope_delta": { "summary": "Eq. (28) itself: the control loss minimized over the controller, with the source law of X, the noise Z and the actuation channel held fixed. The constraint set is exactly the paper's - the minimization is there, in its own words, 'to ensure that L_C reflects the properties of the actuation channel, and does not depend on one's choice of control inputs' - and IsPlant is what holds the channel fixed while the controller varies. Two modelling choices are recorded rather than hidden. Controllers are measurable maps on the ambient space rather than kernels x -> p(c|x), which loses nothing because the space may carry auxiliary randomness independent of X; and the quantity is an sInf rather than a min, so no attainment is assumed. The set of admitted controllers is a parameter, so the unconstrained minimum is the Set.univ case. Examples.Control.minControlLoss_lt_controlLoss_gate exhibits a plant and two controllers whose losses are 0 and log 2, so the minimization is not a formality.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.InformationLimits", "declaration": "AISafetyAtlas.Control.minControlLoss_le_entropy_noise", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 2's inequality, L_C <= H(Z), stated of equation (28)'s minimum rather than of one controller. Purification and the independence step (30) are required at a single admitted controller, since an infimum is capped by any one member.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.InformationLimits AISafetyAtlas.Examples.Control.InformationLimits", "scope_delta": { "summary": "Theorem 2's inequality at a represented policy infimum, not only pointwise in the controller. The pointwise bound transfers because an infimum is capped by one admitted member, and the realized-policy quantity is an upper ceiling on the source kernel infimum. The equality case is separate: it describes a minimizer and remains represented only under an attainment hypothesis for the supplied family P.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.InformationLimits", "declaration": "AISafetyAtlas.Control.minControlLoss_eq_sInf_condMutualInfo", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 3, L_C = I(X' : Z | X, C), stated of equation (28)'s minimum. The pointwise equality controlLoss_eq_condMutualInfo holds at every admitted controller, so the two images coincide and so do their infima for any represented policy family; no minimizer is needed. A common all-kernel realization is not constructed here.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.InformationLimits AISafetyAtlas.Examples.Control.InformationLimits", "scope_delta": { "summary": "Theorem 3 at a represented policy infimum. The two quantities are equal at every admitted controller, so their images under the loss map coincide and hence so do their infima for any represented family - the printed equality is obtainable without a minimizer. The source's all-kernel family is not declared or realized on a common Lean sample space. Generalized from finite alphabets to any measurable space with a zero-or-probability measure.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.InformationLimits", "declaration": "AISafetyAtlas.Control.minControlLoss_eq_sInf_mutualInfo_sub", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 4, L_C = I(X' : X,C,Z) - I(X' : X,C), stated of equation (28)'s minimum, by the same pointwise-to-represented-family transfer as the Theorem 3 record. A common all-kernel realization is not constructed here.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.InformationLimits AISafetyAtlas.Examples.Control.InformationLimits", "scope_delta": { "summary": "Theorem 4 at a represented policy infimum, by the same transfer as the Theorem 3 record: an equality holding at every admitted controller is an equality of the two infima for that family. The source's all-kernel feasible family is not declared or realized on a common Lean sample space. Generalized from finite alphabets to any measurable space with a zero-or-probability measure.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Control.InformationLimits", "declaration": "AISafetyAtlas.Control.minControlLoss_eq_entropy_noise_iff_of_attained", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "touchette-lloyd-2003-preprint" ], "content_source_note": "Theorem 2's equality condition, L_C = H(Z) iff H(Z | X', X, C) = 0, stated of equation (28)'s minimum under the hypothesis that an admitted controller attains it - which is what the paper's min asserts and never exhibits.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d; PFR@7d6404b79b11b1a89dd1b6f997a10b14c208c4ed", "build_command": "lake build AISafetyAtlas.Control.InformationLimits AISafetyAtlas.Examples.Control.InformationLimits", "scope_delta": { "summary": "Theorem 2's equality case for a represented policy infimum. The printed condition describes a minimizer, so this takes 'C is admitted and no admitted controller loses less' as an explicit hypothesis for the supplied family P. It does not identify that minimizer with a minimizer of the source's all-kernel infimum. Without represented-family attainment the equality case stays pointwise as controlLoss_eq_entropy_noise_iff.", "evidence": "docs/provenance/touchette-lloyd-control.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "Oversight", "oversight_reduction_le_budget", "blind_channel_buys_nothing", "budget_is_the_channel_not_the_volume" ], "module": "AISafetyAtlas.Control.OversightBudget", "build_command": "lake build AISafetyAtlas.Control.OversightBudget", "scope_delta": { "summary": "BRIDGE module. Touchette-Lloyd's bound read as an oversight budget, and the only quantitative bridge in the tree: the reduction an oversight regime achieves is at most the printed open-loop maximum openLoopMax F eta plus the mutual information its channel carries about the hazard, provided the noise is independent of the hazard and the reading. Since 2026-10-05 (closure audit) the first term is openLoopMax, fixed by the plant and the noise law and never by the reading; before, it was a parameter Delta-blind under OpenLoopBound, which was vacuous with informative noise when asked on every event and channel-dependent when asked on the reading's fibres. The two terms separate, and only the second is what monitoring moves. A witness is OWED for the two corollaries and the raised gate ceiling records it; see the module header for why a cheap model would be degenerate.", "evidence": "docs/provenance/practitioner-checkers.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Control.entropy_noise_sub_controlLoss", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.controlLoss", "AISafetyAtlas.Control.Purified" ], "application": "Actuation noise bounds control loss: a controller cannot be left more uncertain about the outcome than the channel disturbing it, and the slack is exactly what the outcome fails to reveal about the noise." }, { "atlas_declaration": "AISafetyAtlas.Control.entropyReduction_le_of_openLoopBound", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.OpenLoopBound", "AISafetyAtlas.Control.IsPlant", "AISafetyAtlas.Control.entropyReduction_le_of_condEntropy_ge" ] }, { "atlas_declaration": "AISafetyAtlas.Control.condEntropy_ge_of_openLoopBound", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.OpenLoopBound", "AISafetyAtlas.Control.IsPlant" ] }, { "atlas_declaration": "AISafetyAtlas.Control.controlLoss_le_entropy_noise", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropy_noise_sub_controlLoss" ] }, { "atlas_declaration": "AISafetyAtlas.Control.controlLoss_eq_entropy_noise_iff", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropy_noise_sub_controlLoss" ] }, { "atlas_declaration": "AISafetyAtlas.Control.controlLoss_eq_condMutualInfo", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.controlLoss", "AISafetyAtlas.Control.Purified" ] }, { "atlas_declaration": "AISafetyAtlas.Control.controlLoss_eq_mutualInfo_sub", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.controlLoss", "AISafetyAtlas.Control.Purified" ] }, { "atlas_declaration": "AISafetyAtlas.Control.condEntropy_le_condEntropy_of_forall", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Control.entropyReduction_le_of_condEntropy_ge", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropyReduction" ], "application": "One bit gathered by a controller is worth at most one bit of extra entropy reduction over open-loop control. Feedback is bounded by, and only by, what the sensor actually learned." }, { "atlas_declaration": "AISafetyAtlas.Control.minControlLoss_le_entropy_noise", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.minControlLoss_le", "AISafetyAtlas.Control.controlLoss_le_entropy_noise", "AISafetyAtlas.Control.inputPolicies", "AISafetyAtlas.Control.IsInputPolicy" ], "application": "Control limits: the actuation noise bounds the control loss of the best available controller, not merely of the one in hand. Touchette-Lloyd Theorem 2 at the printed L_C of eq. (28)." }, { "atlas_declaration": "AISafetyAtlas.Control.minControlLoss_eq_sInf_condMutualInfo", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.controlLoss_eq_condMutualInfo" ] }, { "atlas_declaration": "AISafetyAtlas.Control.minControlLoss_eq_sInf_mutualInfo_sub", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.controlLoss_eq_mutualInfo_sub" ] }, { "atlas_declaration": "AISafetyAtlas.Control.minControlLoss_le", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.bddBelow_controlLoss_image" ] }, { "atlas_declaration": "AISafetyAtlas.Control.minControlLoss_nonneg", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.minControlLoss" ] }, { "atlas_declaration": "AISafetyAtlas.Control.measurable_plantOutcome", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.plantOutcome" ] }, { "atlas_declaration": "AISafetyAtlas.Control.isPlant_plantOutcome", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.plantOutcome", "AISafetyAtlas.Control.IsPlant" ] }, { "atlas_declaration": "AISafetyAtlas.Control.minControlLoss_eq_entropy_noise_iff_of_attained", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.minControlLoss_le", "AISafetyAtlas.Control.controlLoss_eq_entropy_noise_iff" ] }, { "atlas_declaration": "AISafetyAtlas.Control.kernelControlLoss_eq_sum", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.kernelMeasure_prod", "AISafetyAtlas.Control.condEntropy_atom_kernelMeasure" ], "application": "Equation (28)'s second displayed line, verbatim: the kernel's control loss is sum over x of p(x) times sum over c of H(X'|x,c) p(c|x). The objective is linear in p(c|x), which is what every other result in the bridge rests on." }, { "atlas_declaration": "AISafetyAtlas.Control.minControlLoss_inputPolicies_eq_kernelMin", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.kernelMinControlLoss_eq", "AISafetyAtlas.Control.minControlLoss_inputPolicies_eq" ], "application": "The representation gap is zero: the infimum over the controllers realizable on one sample space equals the source's minimum over all conditional distributions. Requires finite alphabets, which is the printed setting." }, { "atlas_declaration": "AISafetyAtlas.Control.minControlLoss_inputPolicies_attained", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.bestAction", "AISafetyAtlas.Control.closedFormLoss_le_controlLoss" ], "application": "A minimizer of equation (28), constructed. Deterministic state feedback playing an argmin action at each state. This is what discharges the attainment hypothesis of Theorem 2's equality case, which the source itself never exhibits." }, { "atlas_declaration": "AISafetyAtlas.Control.entropyReduction_le_condEntropy_form", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropyReduction" ], "application": "Lemma 8. Proved with no open-loop assumption at all: the source states it inside the open-loop section but its printed proof uses only that conditioning does not raise entropy." }, { "atlas_declaration": "AISafetyAtlas.Control.entropyReduction_le_iSup_openLoopReduction", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropyReduction_le_condEntropy_form", "AISafetyAtlas.Control.condEntropy_fibre_eq_openLoopEntropy" ], "application": "Theorem 9: no randomized open-loop controller removes more entropy than the best single action. The open-loop model enters as the explicit hypothesis IndepFun C (X,Z), the atlas's rendering of equation (opd)." }, { "atlas_declaration": "AISafetyAtlas.Control.exists_entropyReduction_const_eq_iSup_openLoopReduction", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropyReduction_const", "AISafetyAtlas.Control.openLoopReduction" ], "application": "Theorem 9's attainment clause: for a finite nonempty action alphabet, some constant action reaches the supremum of the pure open-loop reductions. This states the existential conclusion that entropyReduction_const alone did not." }, { "atlas_declaration": "AISafetyAtlas.Control.entropyReduction_le_openLoopMax", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.condEntropy_ge_of_openLoopMax", "AISafetyAtlas.Control.entropyReduction_le_of_condEntropy_ge" ], "application": "Theorem 10 inside the independent-noise rendering of equation (48). Step (50) becomes an instance because each conditional law p(x|c) is ranged over. The realization bridge from arbitrary printed transition kernels is isPurification_purifyMap, with openLoopMax_purifyMap showing the two reduction sets coincide, and maximum attainment is isGreatest_kernelOpenLoopMax; exists_kernelEntropyReduction_le_at_max states Theorem 10 against the realized maximum." }, { "atlas_declaration": "AISafetyAtlas.Control.perfectlyObservable_iff_sensorLoss_eq_zero", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.entropy_eq_zero_iff", "AISafetyAtlas.Control.sensorLoss", "AISafetyAtlas.Control.PerfectlyObservable" ], "application": "Theorem 5, whose proof the source omits as following from well-known properties of entropy. The property is entropy_eq_zero_iff, which is not in the entropy layer and is proved here." }, { "atlas_declaration": "AISafetyAtlas.Control.condMutualInfo_eq_zero_of_sensorLoss_eq_zero", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.sensorLoss" ], "application": "Theorem 6: a perfectly observable state makes the sensor's purification noise uninformative. The source notes the converse fails and it is not claimed." }, { "atlas_declaration": "AISafetyAtlas.Control.mutualInfo_prod_eq_of_sensorLoss_eq_zero", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Control.sensorLoss" ], "application": "Corollary 7. The statement is I(X;C,Z) = I(X;C) with a comma, checked against both the published page and the arXiv TeX source." }, { "atlas_declaration": "AISafetyAtlas.Control.OversightBudget.oversight_reduction_le_budget", "type": "BRIDGE", "source_declarations": [ "Oversight", "blind_channel_buys_nothing", "budget_is_the_channel_not_the_volume" ], "application": "Pricing a monitoring proposal. The reply to 'add more monitoring' is not that it never helps but that it helps by at most the information the channel carries about the hazard, and the bound separates that term from the open-loop maximum, which depends on the plant and the noise law only, never on the reading (given noise independent of hazard and reading). A channel informationally independent of the hazard contributes exactly zero, so an argument for monitoring rests on a mutual information being positive -- a measurable property of the channel rather than a matter of design intent. Read as an upper bound only: a rich channel does not guarantee the regime achieves anything, and the bound says only that it cannot achieve more.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-05", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-oversight-reduction-le-budget.md" } } ] }, "root_import": true, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Source comparison done against the published Physica A text (survey-ref-027); its Theorems 1-11 are numbered identically to the arXiv v2 preprint originally used, and Theorems 2, 3, 4 and 10 match statement for statement. The arXiv TeX source was also read, which settles Corollary 7 as I(X;C,Z) = I(X;C) with a comma and confirms equation (28)'s second displayed line. Recorded in docs/provenance/touchette-lloyd-control.md. The equation (28) representation gap is closed for finite alphabets: kernelMinControlLoss declares the all-kernel object, minControlLoss_inputPolicies_eq_kernelMin identifies it with the realized infimum, and minControlLoss_inputPolicies_attained exhibits a deterministic minimizing policy. Theorem 2's equality case is therefore unconditional at print scope. The actuation-model gap is closed. Section 2 of the published text states purification for ANY non-deterministic channel, before Section 3 and long before the open-loop section, with condition (i) determinism given (c,z) and condition (ii) equation (7); the earlier note claiming the source confines purification to Theorem 2 was wrong and is withdrawn. AISafetyAtlas.Control.Purification proves the assertion the paper only makes: isPurification_purifyMap builds the seed as a random element of the finite space of deterministic channels S x K -> T. Theorem 9 and the step-(50) row are therefore Yes at print scope, via kernelEntropyReduction_le_iSup_kernelOpenLoop and exists_kernelEntropyReduction_dirac_eq_iSup. Equation (48)'s maximum is now proved attained by isGreatest_kernelOpenLoopMax, using the identification of probability measures on a finite state space with points of the standard simplex, linearity of the outcome law in those coordinates, and compactness; exists_kernelEntropyReduction_le_at_max states Theorem 10 against the realized maximum. Theorem 10 is therefore Yes and this source has no Partial rows left. entropyReduction_le_of_openLoopBound remains useful because its conditional-ensemble hypothesis is incomparable with this rendering. Theorem 9's attainment clause is now explicit in exists_entropyReduction_const_eq_iSup_openLoopReduction. Set.univ is not equation (28)'s feasible family: a bare map may read and cancel actuation noise, as Examples.Control.minControlLoss_univ_lt_inputPolicy and not_isInputPolicy_noiseReader show. Lemma 8, Theorems 5, 6, 9 and 10's inequality at kernel scope, and Corollary 7 are proved; Theorems 1 and 11 remain unformalized, as does the source's unproved prose claim L_S <= H(Z_B).", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "information-theoretic control", "information theoretic control" ] }, "tags": [ "control-theory", "information-theory" ], "public": { "group": "Limits of control and regulation", "summary": "Feedback helps only as much as the sensor actually measured.", "use": "Deciding how much monitoring a control loop really needs.", "title": "Information limits on control", "attribution": "Touchette & Lloyd" } }, { "id": "BY-006", "name": "(Anti)codifiability thesis", "paper_reference": "Table 1", "survey_proof_assessment": "NOT_PROVEN_IN_SURVEY", "informal_claim": "The survey groups moral-particularist arguments against complete codification of virtue.", "original_source_refs": [ "survey-ref-028", "survey-ref-029", "survey-ref-030" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "codifiability", "moral particularism", "virtue codification" ] }, "tags": [ "ethics" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; the verdict is unchanged and is now a recorded state rather than boilerplate. NONE OF THE THREE CITED SOURCES IS HELD, and each was checked for a lawful route on 2026-09-13: McDowell, Virtue and Reason, The Monist 62(3), Philosophy Documentation Center, subscription; McKeever and Ridge, The Many Moral Particularisms, Canadian Journal of Philosophy 35(1), subscription; Tsu, Can Virtue Be Codified?, a chapter, subscription. Routes are recorded in the private manifest of 2026-09-13. Sci-Hub is declined and stays declined. ONE THING THE ROW'S OWN WORDING ALREADY SAYS, and it is worth recording before anyone reads those three: the informal claim is that 'the survey GROUPS moral-particularist arguments against complete codification of virtue'. A grouping of arguments is not a theorem, so the expected outcome of reading them is a statement of what would have to be formalized, not a candidate to adjudicate. THE SIX-CORPUS SWEEP on this row returned zero, and a check of the whole tree on 2026-09-13 for 'codifiab' and 'virtue' finds them only in registry.yaml, never in AISafetyAtlas/. Nothing here is close." } }, { "id": "BY-007", "name": "Arrow's impossibility theorem", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "No rank-order voting rule satisfies Arrow's full set of fairness conditions for three or more alternatives.", "original_source_refs": [ "survey-ref-031" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/ChihChengLiang/arrow", "version": "758398779decc66d2830a70b02597b0f22030181", "relationship": "EQUIVALENT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas", "module": "Arrow/Arrow.lean; vendored as AISafetyAtlas.Upstream.Arrow", "declaration": "Impossibility" }, { "framework": "Isabelle/HOL", "repository": "https://www.isa-afp.org/entries/ArrowImpossibilityGS.html", "version": "AFP release 2026-02-06 (SHA256 8174c738b42203100170ff25f3c9fc2c6d16d8556fbaff205c0eaa98a3813da7)", "relationship": "EQUIVALENT", "reproduced": true, "license": "BSD-3-Clause", "build_environment": "makarius/isabelle@sha256:9bd33b183c399327c5d554fc8cde27c29b5d2b20cdc6fe7a604caa3f951018fc", "build_command": "scripts/reproduce_isabelle.sh arrow", "module": "Thys/Arrow_Order.thy; session ArrowImpossibilityGS", "declaration": "Arrow" }, { "framework": "Isabelle/HOL", "repository": "https://www.isa-afp.org/entries/ArrowImpossibilityGS.html", "version": "AFP release 2026-02-06 (SHA256 8174c738b42203100170ff25f3c9fc2c6d16d8556fbaff205c0eaa98a3813da7)", "relationship": "EQUIVALENT", "reproduced": true, "license": "BSD-3-Clause", "build_environment": "makarius/isabelle@sha256:9bd33b183c399327c5d554fc8cde27c29b5d2b20cdc6fe7a604caa3f951018fc", "build_command": "scripts/reproduce_isabelle.sh arrow", "module": "Thys/Arrow_Utility.thy; session ArrowImpossibilityGS", "declaration": "dictator" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.SocialChoice.arrow", "type": "WRAPPER", "source_declarations": [ "AISafetyAtlas.Upstream.Arrow.Impossibility" ] }, { "atlas_declaration": "AISafetyAtlas.SocialChoice.Utility.arrow", "type": "WRAPPER", "source_declarations": [ "AISafetyAtlas.Upstream.Arrow.Impossibility" ], "application": "Aggregation of utility-represented preferences: Arrow's impossibility stated over finite total preorders encoded by lower-contour cardinalities, so a utility profile can be aggregated directly. Representation bridge over the classical Arrow proof." } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "CC Liang's Lean 4 order proof is the canonical mathematical source and is vendored at an immutable Apache-2.0 commit with only module, namespace, visibility, and interface-reducibility changes. The atlas exposes one stable order theorem and proves one utility-representation bridge from it. The two reproduced Isabelle/HOL variants remain provenance and specification references, not additional coverage results.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "mathlib", "isabelle-afp" ], "query_terms": [ "arrow impossibility", "arrow's theorem", "arrow theorem" ] }, "tags": [ "social-choice", "decision-theory" ], "public": { "group": "Aggregation and multi-agent structure", "summary": "No fair way exists to merge people's rankings into one.", "use": "Combining values across people or models.", "title": "Arrow's impossibility", "attribution": "Arrow" } }, { "id": "BY-008", "name": "Impossibility theorems in population ethics", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "No population axiology simultaneously satisfies the cited set of adequacy conditions.", "original_source_refs": [ "survey-ref-032" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "A1 (2026-07-19): AFP CondNormReasHOL (Parfit mere addition / Åqvist E) is RELATED landscape LAND-PE-001 only — not EXACT/EQUIVALENT Arrhenius 2011 (survey-ref-032). See docs/provenance/a1-condnorm-parfit-triage.md.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "isabelle-afp" ], "query_terms": [ "population ethics", "population axiology" ] }, "tags": [ "social-choice", "ethics" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "A1 (2026-07-19) found AFP CondNormReasHOL RELATED only and recorded it as the separate landscape row LAND-PE-001; it is not EXACT or EQUIVALENT to the Arrhenius 2011 impossibility package this row cites. Evidence: docs/provenance/a1-condnorm-parfit-triage.md." } }, { "id": "BY-009", "name": "Impossibility theorems in AI alignment", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "The cited work derives incompatibilities among desiderata for utility-based value alignment.", "original_source_refs": [ "survey-ref-033" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "value alignment", "ai alignment", "utility alignment" ] }, "tags": [ "social-choice", "decision-theory" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "RETRIAGED 2026-09-13, from UNTRIAGED. THE SOURCE IS NOW HELD AND READ AT THE LEVEL THAT DECIDES THIS ROW. Eckersley was fetched from arXiv 2026-09-13 and is pinned as eckersley-arxiv-v1-2019-impossibility-and-uncertainty-theorems-in-ai-value-alignment.pdf, sha256 8a3c7680398e75e0f70147e642be2919cb298f5da0040ec9019837a7cf78937a, 13 pp. VERSION: the banner on the face reads arXiv:1901.00064v3, 5 Mar 2019; the ledger's locator points at the abstract page and the row cited no version. THE DECIDING FACT: the paper states NO numbered theorem, lemma, proposition or corollary of its own - a scan of the full text for those keywords returns only citations and section headings. Its abstract says what it does: 'We show that previously known impossibility theorems can be transformed into uncertainty theorems', and the theorems it transforms are other people's - Arrow's, and Arrhenius's for population axiology. SO THERE IS NOTHING ON THIS ROW TO ADJUDICATE A CANDIDATE AGAINST, and the six-corpus sweep's zero is not a gap: the row's claim is about a body of results that are cited here, not proved here. WHAT THE TREE ALREADY HAS, CHECKED 2026-09-13: AISafetyAtlas.SocialChoice.arrow is Arrow's impossibility in order form, hosted by BY-007 - so one of the two impossibility results this paper leans on is already in the atlas, under a different row. What is not here is Arrhenius's population-axiology impossibility, and nothing in the tree mentions it. A FRESH FORMALIZATION IS OWED AND ITS SHAPE IS NAMED: either Arrhenius's theorem itself, which would be a new row against Arrhenius and not against this paper, or the paper's own transformation of an impossibility into an uncertainty statement, which would first need the transformation stated precisely - print gives it in prose. VOCABULARY NOTE, so the verdict is not read as more than it is: the statability vocabulary has no value for a source that states no theorem of its own. TRIAGED_DISTINCT is the nearest, and its gloss's first clause - candidates were examined and found distinct - is VACUOUS here rather than satisfied, because the sweep examined nothing: there is no printed statement for a candidate to be distinct from. Its second clause, that a fresh formalization is owed, does hold and is what this note above says. UNTRIAGED is wrong, not weaker: its gloss is that nobody has compared the source to the tree, and that comparison is now done. Whether the vocabulary gains a value for this case is a maintainer decision and is not taken here." } }, { "id": "BY-010", "name": "Fairness impossibility theorem", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Common calibration and error-rate fairness criteria cannot all hold when base rates differ, subject to source assumptions.", "original_source_refs": [ "survey-ref-034", "survey-ref-035" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Fairness.RiskAssignment", "declaration": "AISafetyAtlas.Fairness.perfect_prediction_or_equal_base_rates", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-034" ], "content_source_note": "Theorem 1.1 of arXiv:1609.05807v2, the exact characterization: an instance carrying a risk assignment that is calibrated within groups (A) and balanced for the negative (B) and positive (C) classes either allows perfect prediction or has equal base rates. Model and proof are Sections 1.1 and 2; the matrix notation of Section 2 is rendered as finite sums.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Fairness.RiskAssignment AISafetyAtlas.Examples.Fairness.RiskAssignment", "scope_delta": { "summary": "One quantifier, and it is forced. Print's perfect-prediction disjunct reads 'p_sigma equal to 0 or 1 for all sigma'; PerfectPrediction quantifies over the feature vectors somebody has. Section 1.1 never says whether a feature vector may have frequency zero in both groups, and the two readings disagree: at the permissive one the unrestricted sentence is false, refuted in-tree by Examples.Fairness.RiskAssignment.not_print_perfectPrediction on an instance meeting every hypothesis with p = 1/2 at an unpopulated vector; at the restrictive one nothing is lost, since print_perfectPrediction_of_populated recovers print's sentence verbatim and populated_print_perfectPrediction inhabits it. Print's hypotheses are otherwise taken unchanged, including the score upper bound v_b <= 1, which no step uses. Theorem 1.2, the approximate characterization, is a separate record on this row.", "evidence": "docs/provenance/kleinberg-fairness-tradeoffs.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Fairness.RiskAssignment", "declaration": "AISafetyAtlas.Fairness.print_perfectPrediction_of_populated", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-034" ], "content_source_note": "Theorem 1.1's perfect-prediction disjunct at its printed quantifier -- 'p_sigma equal to 0 or 1 for all sigma' -- under the reading of Section 1.1 on which every feature vector belongs to somebody.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Fairness.RiskAssignment AISafetyAtlas.Examples.Fairness.RiskAssignment", "scope_delta": { "summary": "Print's sentence exactly, with the populated-instance hypothesis Section 1.1 leaves unwritten made explicit. It is a hypothesis print does not state, so the record is RELATED rather than a statement match; the hypothesis is inhabited by Examples.Fairness.RiskAssignment.populated.", "evidence": "docs/provenance/kleinberg-fairness-tradeoffs.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Fairness.ApproximateRiskAssignment", "declaration": "AISafetyAtlas.Fairness.approx_perfect_prediction_or_equal_base_rates", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-034" ], "content_source_note": "Theorem 1.2 of arXiv:1609.05807v2, the approximate characterization: an instance carrying a risk assignment satisfying the epsilon-approximate conditions (A'), (B') and (C') of Section 3 satisfies either the f(epsilon)-approximate form of perfect prediction or the f(epsilon)-approximate form of equal base rates. The witness f is print's own, f(x) = sqrt(x) * max(1, 3*sqrt(x) + 3/4), read off the last line of Section 3; exists_slack_function states the existential form over it, with the predicate repairs recorded below. Model and proof are Sections 1.1 and 3.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Fairness.ApproximateRiskAssignment AISafetyAtlas.Examples.Fairness.ApproximateRiskAssignment", "scope_delta": { "summary": "Approximate predicates reconstructed from Section 3's argument, rather than transcribed literally from its displays. Printed (A') forces P = (1-eps)S, where P is a bin's positive-class count and S its total assigned score; this is exact calibration only at eps = 0 and does not make Theorem 1.2 a restatement of Theorem 1.1. ApproxCalibrated instead requires (1-eps)P <= S <= (1+eps)P, changing both the upper sign and the direction to support equation (7). The (B') and (C') displays also use n_t throughout despite group-specific denominators; Lean uses each group's own numerator and interchanges whole groups, consistently with the defined class averages and equation (8). These repairs keep the record RELATED. Section 3 divides by 1 - 2eps + eps^2 - gamma_1 without excluding a nonpositive denominator; the Lean handles that case separately, where the required lower bound already holds. Both classes are explicitly nonempty in each group (0 < mu t < N t), making the printed averages defined. The result holds for eps >= 0, extending print's eps > 0 and supporting the recovery of Theorem 1.1 at zero.", "evidence": "docs/provenance/kleinberg-fairness-tradeoffs.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Fairness.ApproximateRiskAssignment", "declaration": "AISafetyAtlas.Fairness.exists_slack_function", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-034" ], "content_source_note": "Theorem 1.2 in print's own existential form: there is a continuous f with f(x) -> 0 as x -> 0 such that the epsilon-approximate conditions force one of the two f(epsilon)-approximate conclusions.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Fairness.ApproximateRiskAssignment AISafetyAtlas.Examples.Fairness.ApproximateRiskAssignment", "scope_delta": { "summary": "The feature-vector and bin types are quantified inside the existential and so are restricted to universe zero, where the explicit theorem is universe-polymorphic; print says nothing about universes. Continuity and the filter limit at zero express both regularity requirements of Theorem 1.2. The calibration-direction/sign repair and the group-index repairs documented for approx_perfect_prediction_or_equal_base_rates also apply here, as do its nonempty-class assumptions. This is the printed existential form with reconstructed predicates, not a literal transcription of the displayed hypotheses.", "evidence": "docs/provenance/kleinberg-fairness-tradeoffs.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Fairness.ApproximateRiskAssignment", "declaration": "AISafetyAtlas.Fairness.approx_tradeoff_of_score_relative_calibration", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-034" ], "content_source_note": "A competing formalization of Theorem 1.2 using the sign-only orientation displayed for (A') in Section 3: the positive-class mass is within a factor of the total assigned score. The balance predicates retain the group's own numerators and repaired group-index orientation from the main approximate formalization. The continuous witness is scoreRelativeSlack(epsilon) = slack(epsilon / max(1 - epsilon, 1/2)), rather than print's unchanged slack(epsilon).", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Fairness.ApproximateRiskAssignment AISafetyAtlas.Examples.Fairness.ApproximateRiskAssignment", "scope_delta": { "summary": "This is a deliberate sign-only repair of (A'), not a claim that print's display and the main atlas orientation are equivalent. ApproxCalibratedScoreRelative reverses the two WithinFactor arguments, while the (B') and (C') predicates keep the same group-index repairs and nonempty-class hypotheses as approx_perfect_prediction_or_equal_base_rates. This proof rescales epsilon to epsilon/(1 - epsilon) below the 1/2 cutoff; scoreRelativeSlack floors that denominator at 1/2 and handles the global branch where the bound is already at least 1. Its continuous bound is not print's original slack, so the record is RELATED.", "evidence": "docs/provenance/kleinberg-fairness-tradeoffs.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Fairness.ApproximateRiskAssignment", "declaration": "AISafetyAtlas.Fairness.exists_slack_function_score_relative", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-034" ], "content_source_note": "Theorem 1.2's existential conclusion under the competing sign-only orientation of (A'): a continuous f tending to zero at zero is witnessed by scoreRelativeSlack, with the same repaired group-index balance predicates and nonempty-class assumptions as the main approximate formalization.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Fairness.ApproximateRiskAssignment AISafetyAtlas.Examples.Fairness.ApproximateRiskAssignment", "scope_delta": { "summary": "The existential surface is the sign-only competing repair, not a second statement of print's unchanged error bound. Its f is the continuous scoreRelativeSlack rescaling, and its (B') and (C') clauses retain the repaired group-specific averages and indices documented for approx_perfect_prediction_or_equal_base_rates. The same universe-zero existential packaging and explicit nonempty-class hypotheses remain, so this extends the RELATED scope rather than raising it to EXACT.", "evidence": "docs/provenance/kleinberg-fairness-tradeoffs.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "survey-ref-034" ], "content_source_note": "Theorem 1.1 of arXiv:1609.05807v2, the exact characterization: an instance carrying a risk assignment that is calibrated within groups (A) and balanced for the negative (B) and positive (C) classes either allows perfect prediction or has equal base rates. Model and proof are Sections 1.1 and 2; the matrix notation of Section 2 is rendered as finite sums.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "declarations": [ "cannot_have_all_three", "population_escape_of_all_three" ], "module": "AISafetyAtlas.Fairness.Tradeoff", "build_command": "lake build AISafetyAtlas.Fairness.Tradeoff", "scope_delta": { "summary": "The citation form of Theorem 1.1. Print states a disjunction -- perfect prediction OR equal base rates -- and the result is used as its contrapositive, that the three conditions cannot be had together. Those are the same theorem and not the same sentence, and the second is the one a practitioner decides against, because both escapes are properties of the POPULATION and are settled before any classifier exists. The disjunctive form invites a reader to hope one disjunct is arrangeable. Backs atlas-check kind 'fairness'; witnessed at a two-feature population in AISafetyAtlas.Examples.Practitioner.", "evidence": "docs/provenance/kleinberg-fairness-tradeoffs.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Fairness.perfect_prediction_or_equal_base_rates", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.Instance", "AISafetyAtlas.Fairness.RiskAssignment", "AISafetyAtlas.Fairness.Calibrated", "AISafetyAtlas.Fairness.BalancedPositive", "AISafetyAtlas.Fairness.BalancedNegative", "AISafetyAtlas.Fairness.PerfectPrediction", "AISafetyAtlas.Fairness.EqualBaseRates" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.print_perfectPrediction_of_populated", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.PerfectPrediction" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.sum_score_eq_μ", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.Calibrated", "AISafetyAtlas.Fairness.sum_assignedPos" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.negativeScore_eq", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.sum_score_eq_μ", "AISafetyAtlas.Fairness.assigned_eq_add" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.perfect_of_negativeScore_eq_zero", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.Calibrated", "AISafetyAtlas.Fairness.negativeScore" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.approx_perfect_prediction_or_equal_base_rates", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.ApproxCalibrated", "AISafetyAtlas.Fairness.ApproxBalancedPositive", "AISafetyAtlas.Fairness.ApproxBalancedNegative", "AISafetyAtlas.Fairness.ApproxPerfectPrediction", "AISafetyAtlas.Fairness.ApproxEqualBaseRates", "AISafetyAtlas.Fairness.slack", "AISafetyAtlas.Fairness.average_lower_bound" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.exists_slack_function", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.approx_perfect_prediction_or_equal_base_rates", "AISafetyAtlas.Fairness.continuous_slack", "AISafetyAtlas.Fairness.tendsto_slack_zero" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.approx_tradeoff_of_score_relative_calibration", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.ApproxCalibratedScoreRelative", "AISafetyAtlas.Fairness.ApproxBalancedPositive", "AISafetyAtlas.Fairness.ApproxBalancedNegative", "AISafetyAtlas.Fairness.ApproxPerfectPrediction", "AISafetyAtlas.Fairness.ApproxEqualBaseRates", "AISafetyAtlas.Fairness.scoreRelativeSlack", "AISafetyAtlas.Fairness.approx_perfect_prediction_or_equal_base_rates" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.exists_slack_function_score_relative", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.approx_tradeoff_of_score_relative_calibration", "AISafetyAtlas.Fairness.continuous_scoreRelativeSlack" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.average_lower_bound", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.ApproxCalibrated", "AISafetyAtlas.Fairness.ApproxBalancedNegative", "AISafetyAtlas.Fairness.ApproxBalancedPositive", "AISafetyAtlas.Fairness.sum_score_bounds", "AISafetyAtlas.Fairness.baseRate" ] }, { "atlas_declaration": "AISafetyAtlas.Fairness.perfect_prediction_or_equal_base_rates_of_approx", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Fairness.approx_perfect_prediction_or_equal_base_rates", "AISafetyAtlas.Fairness.approxCalibrated_zero_iff", "AISafetyAtlas.Fairness.perfectPrediction_of_approx_zero" ] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "public": { "group": "Aggregation and multi-agent structure", "title": "Fairness criteria that cannot coexist", "attribution": "Kleinberg, Mullainathan & Raghavan", "summary": "A risk score cannot be calibrated within both groups and equally accurate for both classes unless the groups have the same base rate or every case is already certain.", "use": "Choosing between fairness criteria for a scoring system, when the groups being scored differ in how often the outcome occurs." }, "notes": "Graded against survey-ref-034 only. survey-ref-035 (Saravanakumar, arXiv:2007.06024v2) attributes the statistical impossibility theorem to Kleinberg et al. (2016), then states a different causal claim over demographic parity, equalized odds and predictive parity: under its assumptions, a data-generating process satisfying one cannot satisfy either of the others. Figures 1--3 give a d-separation argument and Section 4.4 describes it as a proof, though the result is not isolated as a numbered theorem. The Lean records here formalize Kleinberg--Mullainathan--Raghavan's score-bin theorem, not that causal claim, and cite survey-ref-034 alone as their content source. Formalizing survey-ref-035 would require d-separation and its conditional-independence metrics, which this tree does not carry -- see docs/provenance/d-separation-build-or-depend.md.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "fairness impossibility", "calibration fairness", "equalized odds" ] }, "tags": [ "social-choice" ], "escape_routes": [ { "axis": "RESTRICT_CLASS", "status": "FORMALIZED", "lean": "AISafetyAtlas.Examples.Fairness.RiskAssignment.equalRates_escape", "note": "Restrict the instance class to equal base rates and the three criteria become jointly satisfiable. equalRates_escape is that statement: print's two numeric hypotheses, calibration, both balance conditions, equal base rates, and the failure of perfect prediction, in one conjunction. It is graded FORMALIZED because the restriction is inhabited by a checked model rather than argued, and the last conjunct is what makes this escape distinct rather than the other disjunct in disguise. The pointer named equalRates_conclusion until 2026-09-07; that theorem states the obstruction's own disjunction specialized to the witness, which is not the weakened claim, and two independent reviews caught it. What the escape costs is what the theorem says it costs: the groups must have the same base rate, which is a property of the population and not something a risk assignment can choose." }, { "axis": "RESTRICT_CLASS", "status": "FORMALIZED", "lean": "AISafetyAtlas.Examples.Fairness.RiskAssignment.perfect_escape", "note": "The second disjunct is a second and independent restriction: restrict to perfectly predictable instances. perfect_escape is that statement, in the same shape as equalRates_escape, and its last conjunct is the failure of equal base rates, so neither witness satisfies both disjuncts and neither escape subsumes the other. The cost is that the features must determine the outcome, which is a property of the data rather than of the assignment. PerfectPrediction here is restricted to realized feature vectors; the module header records that reading, and print_perfectPrediction_of_populated is where the unrestricted sentence is recovered." } ] }, { "id": "BY-011", "name": "Limits on preference deduction", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Observed behaviour alone does not identify an agent's reward function and planning process without strong normative assumptions.", "original_source_refs": [ "survey-ref-036" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference", "declaration": "AISafetyAtlas.Preference.exists_planner", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference AISafetyAtlas.Examples.PublicAPI", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference", "declaration": "AISafetyAtlas.Preference.exists_reward", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference AISafetyAtlas.Examples.Registry", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference", "declaration": "AISafetyAtlas.Preference.negPlanner", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference AISafetyAtlas.Examples.Registry", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference", "declaration": "AISafetyAtlas.Preference.lemma_six", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference AISafetyAtlas.Examples.Registry", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference.SourceComplexity", "declaration": "AISafetyAtlas.Preference.Source.ReasonableForF.proposition_seven", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference.SourceComplexity AISafetyAtlas.Examples.SixTargets", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference.SourceComplexity", "declaration": "AISafetyAtlas.Preference.Source.ReasonableForF.proposition_eight", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference.SourceComplexity AISafetyAtlas.Examples.SixTargets", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference.Reasonable", "declaration": "AISafetyAtlas.Preference.ReasonableLanguage.proposition_seven", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference.Reasonable AISafetyAtlas.Examples.PublicAPI", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference.Reasonable", "declaration": "AISafetyAtlas.Preference.ReasonableLanguage.proposition_eight", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference.Reasonable AISafetyAtlas.Examples.PublicAPI", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference.Complexity", "declaration": "AISafetyAtlas.Preference.explanation_complexity_eq_behaviour", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference.Complexity AISafetyAtlas.Examples.PublicAPI", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything. THE SECTION 5.1 LOWER BOUND CLOSED ON 2026-09-20 and this module's narrowing with it. Until then the plain-Kolmogorov lower bound quantified over the manufactured strings encodeExplanation r b and inverted one fixed pairing, where print argues the bound for EVERY compatible pair through the evaluation map; section 20 of docs/provenance/source-coverage-audit.md graded that row Partial and Mixed. evalPair is now print's map - a pair string names a program and a reward, and evaluating runs the one on the other - EvaluatesTo is print's compatibility relation, and behaviour_le_of_evaluatesTo is the bound over every compatible pair with one constant. TWO OBSTACLES, NEITHER VISIBLE FROM PRINT'S SENTENCE. Evaluation is PARTIAL, since a planner need not halt on a reward, so Kolmogorov.plainKMapLe does not apply and plainK_le_of_partrec is its partial-computable-map form. And the program slot is read off the pair as a RUN LENGTH of leading true bits, which makes every program index reachable; reading it through Encodable.encode on bit strings would not, and print's own degenerate pair would then be inexpressible - evaluatesTo_degeneratePair is what inhabits the antecedent. ROUTING, per docs/agent/policy/lean-routing.md: plainK_le_of_partrec is generic mathematics and is KEPT HERE rather than added to AISafetyAtlas.Upstream.KolmogorovMathlib, because that tree is a pinned vendored revision 005ac4c81eefe09642ef561057199d489cd79485 that must not be edited; it is the natural next lemma there and is offered upstream with the rest of that development. NOT SUPERSEDED: explanation_at_least_behaviour, degenerate_explanation_cheap and explanation_complexity_eq_behaviour are a true two-sided fact about encodeExplanation, a different pairing that names no program, and their statements are unchanged. NOT DONE: no ReasonableLanguage is instantiated at plainK, which needs policies as bit strings; and no upper bound is claimed on print's own degenerate pair as a function of the behaviour, since its planner is built from the behaviour and bounding it needs primitive recursiveness of the constant-program map, which the pinned Mathlib does not state - readerPair carries the upper bound instead and is compatible with the same behaviour.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference.Override", "declaration": "AISafetyAtlas.Preference.OverrideModel.OverridesFor", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference.Override AISafetyAtlas.Examples.Registry", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Preference.Override", "declaration": "AISafetyAtlas.Preference.OverrideModel.Overrides", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Preference.Override AISafetyAtlas.Examples.Registry", "scope_delta": { "summary": "Six of the eight numbered results in the source are formalized. Real-valued rather than [-1,1]-valued rewards; the source's Conjecture 9 is stated as a predicate and not proved; Proposition 10 is absent; and no nontrivial reasonable-language witness exists — the only exhibited ReasonableLanguage inhabitant is degenerate, so Propositions 7 and 8 hold there without saying anything.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Preference.exists_planner", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.Planner", "AISafetyAtlas.Preference.Explains" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.exists_reward", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.Planner", "AISafetyAtlas.Preference.Explains" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.consistent_rewards_eq_univ", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.exists_planner" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.neg_twin", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.Planner", "AISafetyAtlas.Preference.Explains" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.greedy_rewardOf", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.greedyAction_max", "AISafetyAtlas.Preference.greedyPlanner", "AISafetyAtlas.Preference.rewardOf" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.Source.ReasonableForF.proposition_seven", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.Source.ReasonableForF.F₁_of_compatible", "AISafetyAtlas.Preference.Source.ReasonableForF.F₂_of_compatible", "AISafetyAtlas.Preference.Source.ReasonableForF.F₃_of_compatible" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.Source.ReasonableForF.proposition_eight", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.Source.ReasonableForF.F₄_F₄", "AISafetyAtlas.Preference.op3_op4" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.ReasonableLanguage.proposition_seven", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.ReasonableLanguage.policy_le_of_compatible", "AISafetyAtlas.Preference.op3_op1_op5", "AISafetyAtlas.Preference.op3_op2_op6", "AISafetyAtlas.Preference.op3_op4_op2_op6" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.ReasonableLanguage.proposition_eight", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.op3_op4", "AISafetyAtlas.Preference.ReasonableLanguage.Compatible" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.behaviour_le_of_evaluatesTo", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.plainK_le_of_partrec", "AISafetyAtlas.Preference.partrec_evalPair" ], "application": "Armstrong and Mindermann section 5.1's lower bound at print's own quantifier: one additive constant bounds the behaviour's plain Kolmogorov complexity by that of EVERY bit string that evaluates to it, where evaluation is a planner program run on a reward. Closed 2026-09-20; the row had been Partial and Mixed because the concrete half inverted one manufactured encoding instead." }, { "atlas_declaration": "AISafetyAtlas.Preference.explanation_at_least_behaviour", "type": "NEW_PROOF", "source_declarations": [ "Kolmogorov.plainKMapLe", "AISafetyAtlas.Preference.decodeBehaviour_encodeExplanation" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.degenerate_explanation_cheap", "type": "NEW_PROOF", "source_declarations": [ "Kolmogorov.plainKMapLe", "AISafetyAtlas.Preference.computable_encodeExplanation" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.explanation_complexity_eq_behaviour", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.explanation_at_least_behaviour", "AISafetyAtlas.Preference.degenerate_explanation_cheap" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.RegretModel.cannot_rule_out_half_maximal_regret", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.exists_planner", "AISafetyAtlas.Wireheading.Corruption.ComplementedClass.halfMaximalRegretBound" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.OverrideModel.mixtureValue_rationalise", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.OverrideModel.mixtureValue", "AISafetyAtlas.Preference.OverrideModel" ] }, { "atlas_declaration": "AISafetyAtlas.Preference.Source.ReasonableForF.theorem_two_conditional", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.Source.ReasonableForF.NotAmongLowestCompatible", "AISafetyAtlas.Preference.Source.ReasonableForF.proposition_seven" ], "application": "Preference inference: observed behaviour does not identify an agent's reward function and planner without normative assumptions. Conditional form of Theorem 2 of Armstrong and Mindermann (NeurIPS 2018)." }, { "atlas_declaration": "AISafetyAtlas.Preference.OverrideModel.rationalise_strictly_better", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Preference.OverrideModel", "AISafetyAtlas.Preference.OverrideModel.regret" ] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "preference deduction", "preference inference", "inverse reinforcement", "planner reward", "reward identifiability", "irrational agents" ] }, "tags": [ "preference-inference", "decision-theory" ], "public": { "group": "Preferences, rewards and incentives", "summary": "Behaviour on its own does not pin down what an agent wants.", "use": "Learning what someone wants from what they do.", "title": "Preference unidentifiability", "attribution": "Armstrong & Mindermann" } }, { "id": "BY-012", "name": "Rice's theorem", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Every nontrivial extensional property of partial computable functions is undecidable.", "original_source_refs": [ "survey-ref-037" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/leanprover-community/mathlib4", "version": "db584cd6d46c92f209a44c0f1c829460d327499d", "module": "Mathlib.Computability.Halting", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas", "declaration": "ComputablePred.rice", "relationship": "EQUIVALENT" }, { "framework": "Lean", "repository": "https://github.com/leanprover-community/mathlib4", "version": "db584cd6d46c92f209a44c0f1c829460d327499d", "module": "Mathlib.Computability.Halting", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas", "declaration": "ComputablePred.rice₂", "relationship": "EXACT" }, { "framework": "Isabelle/HOL", "repository": "https://www.isa-afp.org/entries/Recursion-Theory-I.html", "version": "AFP release 2026-02-06 (SHA256 b5314c859ce3b2876ef01151f394c1a5e6b234b0fc6563698dbb0250c73cd3f8)", "module": "RecEnSet.thy; session Recursion-Theory-I", "declaration": "Rice_2", "relationship": "EQUIVALENT", "reproduced": true, "license": "BSD-3-Clause", "build_environment": "makarius/isabelle@sha256:9bd33b183c399327c5d554fc8cde27c29b5d2b20cdc6fe7a604caa3f951018fc", "build_command": "scripts/reproduce_isabelle.sh rice" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Computability.rice", "type": "WRAPPER", "source_declarations": [ "ComputablePred.rice" ] }, { "atlas_declaration": "AISafetyAtlas.Computability.rice_code_iff", "type": "WRAPPER", "source_declarations": [ "ComputablePred.rice₂" ] }, { "atlas_declaration": "AISafetyAtlas.Verification.rice", "type": "BRIDGE", "source_declarations": [ "ComputablePred.rice₂" ], "application": "Verification of program behaviour: Rice's theorem restated for properties of partial input/output behaviour over Mathlib program codes. The interface the behavioural-verification results below reduce through.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-07-19", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-by-012-agentbehavior.md" } } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "REVIEWED", "interpretation_review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-07-19", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-by-012-agentbehavior.md" }, "notes": "Mathlib v4.33.0 provides an equivalent extensional-function theorem and an exact code-set characterization. The atlas keeps those canonical wrappers and derives one semantic-to-code verification bridge for properties stated on partial input/output behavior; it makes no claim about a particular AI system. Isabelle Rice_2 independently verifies the ordinary undecidability result. Reproduced Rice_1 adds an explicit one-reduction and Rice_3 adds a c.e.-set interface, but neither is an additional survey coverage result or a current Lean migration target. Downstream consumer of Verification.rice: AISafetyAtlas.Verification.AgentBehavior.no_behavioral_safety_verifier models encoded agents as Mathlib program codes and proves no total BehavioralSafetyVerifier exists for a nontrivial I/O safety specification (CT-4 / R6-8). Melo et al. (arXiv:2408.08995) apply Rice to a comparable agent/program and non-trivial I/O-judge decision problem; the atlas formalizes the Rice packaging, not that paper’s full narrative or architecture proposals. Maintainer bridge review 2026-07-19 (REVIEWED): statement and scoped AI-safety interpretation accepted (see interpretation_review evidence). Related literature: atlas-ref-melo-2024; related literature docs/guide/related-literature.md. Bridge REVIEWED (2026-07-19) applies to the AI-facing bridges Verification.rice and AgentBehavior.no_behavioral_safety_verifier (statement + scoped interpretation); it is not a re-audit of Mathlib Rice as novel AI-safety content. See docs/interpretation-reviews/review-by-012-agentbehavior.md and docs/releases/v0.2.md. MOVED 2026-09-11: AISafetyAtlas.Verification.AgentBehavior.no_behavioral_safety_verifier used to be listed here. It is a CONSUMER of this row's Rice packaging, not a rendering of Rice's statement, and its own statement source is Melo, Maximo, Soma and Castro (arXiv:2408.08995v1). It now belongs to LAND-VERIF-AGENTBEHAVIOR-001, which is the statement row; this stays the row for the proof route. Listing it here was also why the module-ledger check reported that module as unrowed - the check reads formalizations module fields, and this row's modules are Mathlib's and Isabelle's. See section 24 of docs/provenance/source-coverage-audit.md.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "agda-stdlib" ], "query_terms": [ "rice theorem", "rice's theorem", "rice_" ] }, "tags": [ "computability", "verification" ], "public": { "group": "Limits of verification", "summary": "You cannot build a tool that reads any code and tells you what it does.", "use": "Why no tool can vet arbitrary AI code.", "title": "Rice's theorem", "attribution": "Rice" } }, { "id": "BY-013", "name": "Unprovability", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Sufficiently expressive consistent formal systems contain true statements they cannot prove, under the source's hypotheses.", "original_source_refs": [ "survey-ref-001" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/FormalizedFormalLogic/Foundation", "version": "30a16ffa93d79d73ab4d02427fa00f50e039bf29", "module": "Foundation.FirstOrder.Incompleteness.First", "declaration": "LO.FirstOrder.Arithmetic.exists_true_but_unprovable_sentence", "relationship": "EQUIVALENT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; Foundation@30a16ffa93d79d73ab4d02427fa00f50e039bf29", "build_command": "lake build AISafetyAtlas" }, { "framework": "Lean", "repository": "https://github.com/FormalizedFormalLogic/Foundation", "version": "30a16ffa93d79d73ab4d02427fa00f50e039bf29", "module": "Foundation.FirstOrder.Incompleteness.Second", "declaration": "LO.FirstOrder.Arithmetic.consistent_unprovable", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; Foundation@30a16ffa93d79d73ab4d02427fa00f50e039bf29", "build_command": "lake build AISafetyAtlas" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Logic.godel_first_incompleteness", "type": "WRAPPER", "source_declarations": [ "LO.FirstOrder.Arithmetic.exists_true_but_unprovable_sentence" ] }, { "atlas_declaration": "AISafetyAtlas.Logic.godel_second_incompleteness", "type": "WRAPPER", "source_declarations": [ "LO.FirstOrder.Arithmetic.consistent_unprovable" ] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Gödel's first incompleteness theorem (exists_true_but_unprovable_sentence) is reproduced from FormalizedFormalLogic/Foundation at the pinned revision and exposed via the thin atlas wrapper AISafetyAtlas.Logic.godel_first_incompleteness; this is the BY-013 statement match (EQUIVALENT). The theorem is the classical result for any concrete Δ₁-definable, 𝚺₁-sound arithmetic theory interpreting 𝗥₀ — not an abstract axiomatized skeleton. Gödel's second incompleteness theorem (consistent_unprovable: T ⊬ Con(T) for consistent theories interpreting 𝗜𝚺₁) is a distinct companion result, exposed via AISafetyAtlas.Logic.godel_second_incompleteness and recorded here as RELATED (not the BY-013 first-incompleteness claim, so not double-counted). All three atlas wrappers depend only on [propext, Classical.choice, Quot.sound].", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "mathlib", "isabelle-afp", "hol-light" ], "query_terms": [ "gödel incompleteness", "goedel incompleteness", "unprovability" ] }, "tags": [ "provability-logic" ], "public": { "group": "Limits of self-knowledge and reflection", "summary": "Some truths cannot be proved. A system cannot even prove it is consistent.", "use": "Systems asked to verify themselves.", "title": "Gödel incompleteness I & II", "attribution": "Gödel" } }, { "id": "BY-014", "name": "Undecidability", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "No effective procedure decides every instance of the relevant computation or decision problem.", "original_source_refs": [ "survey-ref-002", "survey-ref-038" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/leanprover-community/mathlib4", "version": "db584cd6d46c92f209a44c0f1c829460d327499d", "module": "Mathlib.Computability.Halting", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas", "declaration": "ComputablePred.halting_problem_re", "relationship": "RELATED" }, { "framework": "Lean", "repository": "https://github.com/leanprover-community/mathlib4", "version": "db584cd6d46c92f209a44c0f1c829460d327499d", "module": "Mathlib.Computability.Halting", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas", "declaration": "ComputablePred.halting_problem", "relationship": "EXACT" }, { "framework": "Lean", "repository": "https://github.com/leanprover-community/mathlib4", "version": "db584cd6d46c92f209a44c0f1c829460d327499d", "module": "Mathlib.Computability.Halting", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas", "declaration": "ComputablePred.halting_problem_not_re", "relationship": "RELATED" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Computability.halting_re", "type": "WRAPPER", "source_declarations": [ "ComputablePred.halting_problem_re" ] }, { "atlas_declaration": "AISafetyAtlas.Computability.halting_problem", "type": "WRAPPER", "source_declarations": [ "ComputablePred.halting_problem" ] }, { "atlas_declaration": "AISafetyAtlas.Computability.nonhalting_not_re", "type": "WRAPPER", "source_declarations": [ "ComputablePred.halting_problem_not_re" ] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Mathlib v4.33.0 formalizes recursive enumerability, noncomputability, and non-r.e. complement results for the fixed-input halting predicate.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light" ], "query_terms": [ "halting problem", "undecidability", "undecidable" ] }, "tags": [ "computability" ], "public": { "group": "Limits of verification", "summary": "You cannot build a tool that tells you if a program will ever stop.", "use": "The root of most impossibilities here.", "title": "The halting problem", "attribution": "Turing; Church" } }, { "id": "BY-015", "name": "Chaitin incompleteness", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "A finitely specified formal theory cannot prove arbitrarily high lower bounds on Kolmogorov complexity.", "original_source_refs": [ "survey-ref-039" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/AlexeyMilovanov/kolmogorov-complexity-lean", "version": "005ac4c81eefe09642ef561057199d489cd79485", "module": "KolmogorovMathlib.Complexity.Chaitin", "declaration": "FormalSystem.chaitinIncompleteness", "relationship": "EQUIVALENT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.31.0; mathlib@fabf563a7c95a166b8d7b6efca11c8b4dc9d911f (upstream pinned checkout)", "build_command": "scripts/reproduce_chaitin.sh" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Logic.chaitin_incompleteness", "type": "WRAPPER", "source_declarations": [ "Kolmogorov.FormalSystem.chaitinIncompleteness" ] }, { "atlas_declaration": "AISafetyAtlas.Logic.chaitin_bound", "type": "WRAPPER", "source_declarations": [ "Kolmogorov.FormalSystem.chaitinBound" ] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "External Lean formalization of Chaitin's incompleteness is reproduced at the pinned revision (EQUIVALENT) and exposed via thin atlas wrappers under AISafetyAtlas.Logic. The proof body is vendored under AISafetyAtlas.Upstream.KolmogorovMathlib for Lean module compatibility (Apache-2.0, same pin). The development uses an abstract sound r.e. FormalSystem, not a concrete arithmetic theory. Classical Gödel first and second incompleteness (concrete arithmetic theories) are covered separately under BY-013 via FormalizedFormalLogic/Foundation; the earlier Kritchman–Raz second-incompleteness skeleton has been retired.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "chaitin incompleteness", "kolmogorov complexity incompleteness" ] }, "tags": [ "algorithmic-information", "provability-logic" ], "public": { "group": "Limits of self-knowledge and reflection", "summary": "A system cannot prove that anything is much more complex than itself.", "use": "Limits on what a system can prove about complexity.", "title": "Chaitin incompleteness", "attribution": "Chaitin" } }, { "id": "BY-016", "name": "Undefinability", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Truth for a sufficiently expressive formal language is not definable within that language under the source's hypotheses.", "original_source_refs": [ "survey-ref-040" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/FormalizedFormalLogic/Foundation", "version": "30a16ffa93d79d73ab4d02427fa00f50e039bf29", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; Foundation@30a16ffa93d79d73ab4d02427fa00f50e039bf29", "build_command": "lake build AISafetyAtlas", "module": "Foundation.FirstOrder.Incompleteness.Tarski", "declaration": "LO.FirstOrder.Arithmetic.undefinability_of_truth", "relationship": "EQUIVALENT" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Logic.tarski_undefinability", "type": "WRAPPER", "source_declarations": [ "LO.FirstOrder.Arithmetic.undefinability_of_truth" ] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Tarski's undefinability of truth is reproduced from FormalizedFormalLogic/Foundation at the pinned revision and exposed via AISafetyAtlas.Logic.tarski_undefinability. The statement is the classical arithmetic form (no arithmetic truth predicate for the standard model), classified EQUIVALENT to the survey informal claim. A related no-Tarski-predicate lemma for consistent IΣ₁-extensions is upstream but not separately counted.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "hol-light" ], "query_terms": [ "tarski undefinability", "undefinability of truth", "undefinability" ] }, "tags": [ "provability-logic" ], "public": { "group": "Limits of self-knowledge and reflection", "summary": "A language cannot define truth for its own sentences.", "use": "Systems that score their own output.", "title": "Tarski undefinability", "attribution": "Tarski" } }, { "id": "BY-017", "name": "Unsurveyability", "paper_reference": "Table 1", "survey_proof_assessment": "NOT_PROVEN_IN_SURVEY", "informal_claim": "Some proofs may exceed feasible human survey or comprehension even when mechanically checkable.", "original_source_refs": [ "survey-ref-041" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "unsurveyability", "surveyability of proof" ] }, "tags": [ "interpretability" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged and now a recorded state. THE SOURCE IS NOT HELD: Bassler, The Surveyability of Mathematical Proof: A Historical Perspective, Synthese 148(1), Springer, subscription. Checked for a lawful route 2026-09-13 and recorded in the private manifest of 2026-09-13. Sci-Hub is declined. A CHECK OF THE TREE, 2026-09-13: 'surveyab' occurs only in registry.yaml, never in AISafetyAtlas/. The six-corpus sweep returned zero. WHAT WOULD HAVE TO BE DECIDED FIRST, and cannot be before the paper is read: whether print's surveyability is a claim about proof LENGTH, about human working memory, or about the historical practice of checking - the three have different formal shapes and only the first is obviously statable. The row's informal claim - some proofs exceed feasible human survey even when mechanically checkable - is compatible with all three." } }, { "id": "BY-018", "name": "Unlearnability", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Learning can be undecidable or computationally infeasible for specified concept classes and learning models.", "original_source_refs": [ "survey-ref-042", "survey-ref-043", "survey-ref-044" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "unlearnability", "learnability undecidable", "pac learnability" ] }, "tags": [ "learning-theory", "computability" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged. ONE OF THE THREE SOURCES IS NOW HELD: Valiant, A Theory of the Learnable, fetched 2026-09-13 and pinned as valiant-cacm-1984-a-theory-of-the-learnable.pdf, sha256 5ab696a5f2afe34fc184872a11e7fe120782094b3b4810a0f1ac0a21f99ece59, 9 pp., the CACM research-contributions typesetting with David Waltz as editor on the face. It carries Claims 1 to 4 and no other numbered statement. THE SOURCE THAT CARRIES THIS ROW'S CLAIM IS NOT HELD: Ben-David, Hrubes, Moran, Shpilka and Yehudayoff, Learnability can be undecidable, Nature Machine Intelligence 1 - Springer Nature, subscription, and no arXiv deposit was located on 2026-09-13. Reyzin's Nature news item is also behind the same subscription. The row's informal claim is the undecidability one, so the verdict cannot move on Valiant alone. WHAT THE TREE HAS NEXT DOOR, CHECKED 2026-09-13: PAC-shaped material in AISafetyAtlas.Verification and AISafetyAtlas.Learning, and learnability language in the Wireheading CRMDP modules - none of which is about undecidability of learnability, and none of which cites any of these three." } }, { "id": "BY-019", "name": "Unpredictability of rational agents", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "No predictor can universally forecast the behaviour of rational agents in the cited strategic settings.", "original_source_refs": [ "survey-ref-045", "survey-ref-046" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "predicting rational agents", "rational agent unpredictability" ] }, "tags": [ "multi-agent", "decision-theory" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged. NEITHER SOURCE IS HELD. Foster and Young's JHU working paper 423 has a RePEc record and the JHU economics working-paper server as its route; Koppl and Rosser is in Metroeconomica 53(3), Wiley, subscription. Both checked 2026-09-13, recorded in the private manifest of 2026-09-13. WHAT IS ALREADY IN THE TREE AND MUST NOT BE CONFUSED WITH THIS ROW: the atlas carries self-prediction and diagonalization results - Brandenburger-Keisler through CLM-LAWVERE-CCC-001, and the Wolpert inference-device impossibility at BY-024. Those are about a predictor predicting ITSELF or an interacting pair; print here is about predicting a RATIONAL OPPONENT in a strategic setting, which is a different obstruction and may not reduce to a diagonal argument at all. That distinction is the reason this row is not simply folded into BY-024, and it is a reading task." } }, { "id": "BY-020", "name": "No Free Lunch — supervised learning", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Averaged uniformly over target functions, learning algorithms have no universal performance advantage.", "original_source_refs": [ "survey-ref-018" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Learning", "declaration": "AISafetyAtlas.Learning.no_free_lunch_supervised", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "wolpert-1996-lack" ], "content_source_note": "The formalized content (uniform target average, off-training-set error learner-independence, homogeneous loss) is from Wolpert's 'Lack' paper. original_source_refs mirrors the survey's own citation (survey-ref-018 = the companion 'Existence' paper, which proves the converse); see docs/provenance/lean-wolpert-nfl.md.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Learning AISafetyAtlas.Examples.PublicAPI AISafetyAtlas.Examples.NFLConcrete", "scope_delta": { "summary": "Finite-domain supervised off-training-set uniform-averaging core under homogeneous loss only; cross-validation as a meta-algorithm, general loss, and stochastic learners from Wolpert 1996 are not mechanized. Distinct from the SSBD PAC no-free-lunch theorem (LAND-NFL-001).", "evidence": "docs/provenance/lean-wolpert-nfl.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Learning.no_free_lunch_supervised", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.sum_pointLoss_off_training", "AISafetyAtlas.Learning.aggregateOffTrainingLoss_eq" ], "application": "Supervised learning: for finite domains and a fixed training set, any two learners have equal off-training-set error averaged uniformly over targets. Finite-domain core of Wolpert (1996)." }, { "atlas_declaration": "AISafetyAtlas.Learning.lossConfig_sum_learner_indep", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.HomogeneousLoss", "AISafetyAtlas.Learning.lossConfig" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.ots_error_distribution_learner_indep", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.lossConfig_sum_learner_indep" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.homogeneous_of_learner_indep", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.HomogeneousLoss", "AISafetyAtlas.Learning.lossConfig" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.homogeneous_iff_learner_indep", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.homogeneous_of_learner_indep", "AISafetyAtlas.Learning.lossConfig_sum_learner_indep" ] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Lean NEW_PROOF (2026-07-19): AISafetyAtlas.Learning.no_free_lunch_supervised formalizes the finite-domain supervised off-training-set uniform-averaging core of Wolpert 1996 — for finite X,Y and fixed training domain S, any two learners A,B : (S→Y)→(X→Y) have equal sum over all targets f of OTS 0-1 loss (closed form |Sᶜ|·(|Y|−1)·|Y|^{|X|−1}). Relationship is RELATED, not EXACT/EQUIVALENT: full 1996 development (cross-validation as meta-algorithm, general loss, stochastic learners) is not mechanized. CT-2: AFP No_Free_Lunch_ML / SSBD PAC NFL remains DISTINCT (LAND-NFL-001). Optimization twin is BY-021 / no_free_lunch. Content source: Wolpert 'The Lack of A Priori Distinctions Between Learning Algorithms' (Neural Comput. 1996, 8(7):1341-1390, doi 10.1162/neco.1996.8.7.1341); survey-ref-018 mirrors the survey's own citation of the companion 'Existence' paper (pp. 1391-1420, doi ...1391), which proves the converse — see provenance. Provenance: docs/provenance/lean-wolpert-nfl.md. ai_interpretation_status stays HUMAN_REVIEW. CT-10 does not land on this row. The sharp closed-under-permutation characterization (AISafetyAtlas.Learning.Sharp) is filed on BY-021, because both of its sources — Igel-Toussaint 2004 and Schumacher-Vose-Whitley 2001 — are optimization papers quantifying over non-repeating black-box *search* algorithms, not over learners. An earlier revision filed those records here and described the atlas as narrowing the algorithm axis to non-adaptive schedules; both were wrong, the second since nfl_adaptive_of_permInvariant proved sufficiency over the printed adaptive class.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "isabelle-afp" ], "query_terms": [ "no free lunch", "supervised learning" ] }, "candidate_formalizations": [], "tags": [ "learning-theory" ], "public": { "group": "Learning and generalization", "summary": "No learner beats another once you average over all possible tasks.", "use": "Making “our method generalizes” name its assumption.", "title": "No free lunch — supervised learning", "attribution": "Wolpert" } }, { "id": "BY-021", "name": "No Free Lunch — optimization", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Averaged uniformly over objective functions, optimization algorithms have equal aggregate performance.", "original_source_refs": [ "survey-ref-019" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Learning", "declaration": "AISafetyAtlas.Learning.no_free_lunch", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Learning AISafetyAtlas.Examples.PublicAPI AISafetyAtlas.Examples.NFLConcrete", "scope_delta": { "summary": "Finite-domain non-adaptive core plus a deterministic adaptive no-revisit strengthening; adaptive query trees, stochastic algorithms, and time-varying objectives from Wolpert-Macready 1997 are not mechanized. The sharp closed-under-permutation characterization is proved over the printed adaptive class in AISafetyAtlas.Learning.Sharp (CT-10), filed on this row.", "evidence": "docs/provenance/lean-wolpert-nfl.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Learning.Sharp", "declaration": "AISafetyAtlas.Learning.nfl_iff_permInvariant", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "igel-toussaint-2003", "igel-toussaint-2004", "schumacher-vose-whitley-2001" ], "content_source_note": "CT-10. The closed-under-permutation characterization, both directions: aggregate performance is schedule-independent iff the weighting of objectives is invariant under permutations of the search domain. The printed condition is that the weight is constant on each basis class, and by their Lemma 1 a basis class is the permutation orbit, so it is the same condition as PermInvariant. The published 2004 JMMA paper has been read. Its Theorem 5, 'non-uniform sharpened NFL', confirms the assessment recorded here: the condition is that the weight is constant on each basis class, its Lemma 1(2) identifies a basis class with a permutation orbit, and the conclusion is quantified over any two non-repeating black-box (hence adaptive) algorithms, any k, any performance measure, AND any m in {1,...,|X|}. Schumacher-Vose-Whitley 2001 has also been read; its sharpened NFL is 'a No Free Lunch result holds over F iff F is closed under permutation', over the same adaptive algorithm class, and its NFL4 explicitly notes that weighted overall measures are not generally subject to NFL except under equal weighting.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Learning.Sharp AISafetyAtlas.Examples.Learning.Sharp", "scope_delta": { "summary": "Not EXACT or EQUIVALENT, but no longer on the algorithm axis. The printed theorem quantifies over non-repeating black-box search algorithms, which are ADAPTIVE, and nfl_adaptive_of_permInvariant now proves the sufficiency half over exactly that class: AdaptiveRule together with the no-revisit hypothesis. nfl_of_permInvariant is the fixed-schedule special case, kept because it is the form the rest of the atlas consumes. What still separates this from the source is the weight axis in the atlas's favour, and time-varying objectives, which remain out of scope. The paper's stochastic algorithms are NO LONGER out of scope: see the nfl_stochastic_of_permInvariant and mixtureTrace records on this row, where the printed quantifier is met at Droste-Jansen-Wegener's own definition. The weight axis is the widening: the atlas statement holds for an arbitrary real weight, including signed ones, where the source takes a probability distribution. The necessity half assumes less than the source and is therefore stronger. Note that quantifying over every sample length is not a sharpening: the source quantifies over m as well, as the published text confirms.", "evidence": "docs/provenance/lean-wolpert-nfl.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Learning.Sharp", "declaration": "AISafetyAtlas.Learning.nfl_stochastic_of_permInvariant", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "igel-toussaint-2004", "droste-jansen-wegener-2002" ], "content_source_note": "Igel-Toussaint state Theorem 1 for 'any two (deterministic or stochastic, cf. [1]) algorithms a and b', citing Droste-Jansen-Wegener on randomized search heuristics. Everything else on this row quantifies over AdaptiveRule, which is deterministic, so the atlas covered a subclass of the printed one. StochasticRule closes that: a rule reads a choice at each step alongside the costs it has seen, and induced fixes the choice sequence to give a deterministic rule DEFINITIONALLY. The theorem is that two stochastic non-repeating rules over possibly different choice alphabets score alike against a permutation-invariant weighting, provided their choice weights carry the same total mass. no_free_lunch_stochastic_of_sharp is the printed uniform case with both masses one.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Learning.Sharp AISafetyAtlas.Examples.Learning.Sharp", "scope_delta": { "summary": "The modelling choice is the content. Randomness is NOT modelled as a weighted family of deterministic rules: that would prove only that No Free Lunch is preserved under mixtures, and would leave the residual it is meant to close, since nothing would say the source's algorithms ARE such mixtures — the same shape as the openLoopMax family residual on BY-005, which needed isPurification_purifyMap to close. Here the randomness sits where a randomized heuristic puts it, at the step, so induced needs no representation theorem. An arbitrary weight on choice SEQUENCES allows correlated choices, so independent per-step randomization is a special case and so is the abstract mixture picture, by taking the choice alphabet to index rules. The finite-choice residual recorded when this landed is CLOSED, at the primary source. Droste-Jansen-Wegener, Theoretical Computer Science 287 (2002) 131-144, is Igel-Toussaint's reference [1]; its Theorem 1 is stated for 'an arbitrary (randomized or deterministic) search heuristic' and its randomized half reads, at p. 134: 'The number of different deterministic search strategies is finite. Let m be its number. A randomized search strategy is a probability distribution p = (p1, ..., pm) and chooses the ith deterministic strategy with probability pi. ... the expected cost of a randomized search heuristic is the weighted average of the cost of the deterministic search heuristics. Since all deterministic search heuristics have the same cost, this also holds for all randomized search heuristics.' So the mixture model is DJW's definition rather than a rendering of one, finiteness of the strategy set is DJW's own observation rather than an atlas restriction, and the final sentence is mixtureTrace_eq_sum_mul, so the atlas proof follows the printed one step for step. Igel's own later survey (igel-2014) says the same in secondary form. mixtureTrace is that equation and nfl_mixture_of_permInvariant is the theorem over it, so the printed quantifier is met at the source's own definition. A is finite because X and Y are, and surjective_induced_playChoice records that the choice-sequence form reaches all of it. mixtureTrace_pointMass is the survey's remark that deterministic algorithms are the degenerate distributions. Nonempty on the choice alphabet remains a hypothesis print does not state, and applies only to the choice-sequence form. Examples.Learning.coinRule_realizations_differ checks the quantifier is not decoration — two choice sequences induce rules that open at different points — and nfl_coinRule_vs_probeRule scores that coin against the branching deterministic probeRule.", "evidence": "docs/provenance/lean-wolpert-nfl.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Learning.Sharp", "declaration": "AISafetyAtlas.Combinatorics.card_closedUnderPermutation_nonempty", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "igel-toussaint-2004", "igel-toussaint-2003" ], "content_source_note": "Igel-Toussaint's Theorem 3: the non-empty subsets of Y^X that are closed under permutation number 2^C(|X|+|Y|-1,|X|) - 1, out of 2^(|Y|^|X|) - 1 non-empty subsets in all. They state it citing their own earlier paper, where the proof is given, so this transcribes a printed count rather than supplying an argument the source lacks. The row pairs with nfl_adaptive_iff_permInvariant: that theorem says No Free Lunch holds exactly on the permutation-closed priors, and this one says almost no prior is permutation-closed. card_closedUnderPermutation is the count without the non-emptiness restriction, and fraction_closedUnderPermutation is the printed fraction, in Q.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Learning.Sharp AISafetyAtlas.Examples.Learning.Sharp", "scope_delta": { "summary": "Same scope as print, at print's own finite alphabets. The content is closedUnderPermutationEquivSet: a permutation-closed set is a union of orbits, an orbit is a basis class by the source's Lemma 1 (already in the tree as basisClass_histogram_eq_permOrbit), and a basis class is determined by the multiset of cost values an objective takes. So the orbits are exactly Sym Y |X|, and Mathlib's Sym.card_sym_eq_choose supplies the tally. spectrum is that multiset, surjective_spectrum shows none is missed, and preimage_image_spectrum is the recovery step. Counts are stated as card + 1 = 2^N so that no truncated natural subtraction appears in a hypothesis-free claim. NOT covered, and split out as its own audit row rather than absorbed here: the asymptotic claim that the fraction converges to zero double exponentially fast for |Y| > e|X|/(|X|-e), which is the point the section exists to make and needs an estimate on binomial coefficients that these counts do not supply.", "evidence": "docs/provenance/lean-wolpert-nfl.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Learning.Sharp", "declaration": "AISafetyAtlas.Combinatorics.exists_perm_rel_not_iff", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "igel-toussaint-2004" ], "content_source_note": "Igel-Toussaint's Theorem 4, the argument the section exists to make: a non-trivial neighbourhood relation on the search space is not invariant under permutations of that space, so the characterization on this row is met by nothing that carries structure. Print defines the object at p. 318: 'A neighborhood relation on X is a symmetric function n: X x X -> {0,1}', non-trivial iff some pair of DISTINCT points neighbours and some pair of distinct points does not. exists_perm_rel_not_iff is that statement and its equation (6). forall_rel_of_permInvariant carries the proof: one instance of the relation at a pair of distinct points forces it at every pair of distinct points, by two-transitivity - a permutation assembled from two transpositions. rel_diag_iff_of_permInvariant is the diagonal half: an invariant relation is all-or-nothing off the diagonal and all-or-nothing on it, the two independently. No count is claimed or proved. forall_adj_or_forall_not_adj_of_permInvariant and exists_perm_adj_not_iff are the SimpleGraph corollaries, kept because that is the form a reader looking for graph automorphisms would search for. Examples.Learning.oneEdge_not_permInvariant is the smallest instance, one edge on three points. Examples.Learning.loopedEdge_not_permInvariant is that edge plus every self-loop: legal under print's definition, non-trivial in print's own sense, and not a SimpleGraph, so it is the witness the bare-relation form reaches and the graph form does not. Examples.Learning.top_permInvariant and Examples.Learning.bot_permInvariant occupy both branches of the graph dichotomy. Examples.Learning.eq_permInvariant is the diagonal, which is invariant but TRIVIAL in print's sense - it relates no two distinct points - so Theorem 4 says nothing about it; it is recorded to show the diagonal is a degree of freedom forall_rel_of_permInvariant does not constrain.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Learning.Sharp AISafetyAtlas.Examples.Learning.Sharp", "scope_delta": { "summary": "Wider than print on two axes. forall_rel_of_permInvariant is stated for an arbitrary binary relation on an arbitrary type: no finiteness, no symmetry, no irreflexivity, and nothing about search in the statement. Print's neighbourhood relation is symmetric; the proof never uses symmetry, so the hypothesis is dropped rather than assumed. The search reading is a corollary. Modelling print's relation as a SimpleGraph would be NARROWER than print, since SimpleGraph is symmetric AND irreflexive whereas print asks only for symmetry and constrains n(x,x) not at all. Examples.Learning.loopedEdge_not_permInvariant is a relation print's definition admits and SimpleGraph cannot express, an edge plus every self-loop, so the graph form cannot be instantiated at it. That is applicability rather than strength: loopedEdge is symmetric, and for symmetric r the graph 'fun a b => a != b and r a b' carries both non-triviality clauses, so the conclusion stays derivable from the graph form. The axis that buys strength is symmetry, witnessed by Examples.Learning.arrowRel_not_permInvariant, where print's theorem cannot be posed at all since no symmetric relation on a two-element type meets both non-triviality clauses. Note print's next sentence, which cuts against the reading here: 'There are only two trivial neighborhood relations, either every two points are neighbored or no points are neighbored.' Print counts off-diagonal behaviour and treats n(x,x) as immaterial rather than as a parameter, so the free diagonal is an inference from a definition that omits it, not something print asserts.", "evidence": "docs/provenance/source-coverage-audit.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Learning.Sharp", "declaration": "AISafetyAtlas.Learning.permInvariant_of_nfl", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "igel-toussaint-2004" ], "content_source_note": "The necessary half of CT-10, and the half that is genuinely stronger than the source: permutation-invariance is forced by schedule-independence at the single sample length |X|, over non-adaptive schedules only, whereas the printed theorem assumes it for every sample length and for the larger class of adaptive non-repeating algorithms. A weaker hypothesis for the same conclusion.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Learning.Sharp AISafetyAtlas.Examples.Learning.Sharp", "scope_delta": { "summary": "Half of the printed iff, with a strictly weaker hypothesis than the source uses: only full-length schedules are probed, and only non-adaptive ones. Assuming NFL over a smaller class of algorithms makes this the stronger of the two halves. The hypothesis is weaker on three axes rather than one: fixed schedules rather than the whole algorithm class, a single sample length, and indicator measures only, which is exactly Igel-Toussaint's delta(k, c(Y)) form. Since the class appears in the hypothesis, a smaller one makes the theorem stronger. The sufficiency direction is now proved over the printed adaptive class by nfl_adaptive_of_permInvariant; time-varying objectives remain out of scope; the stochastic algorithms do not, and are covered by nfl_mixture_of_permInvariant on this same row.", "evidence": "docs/provenance/lean-wolpert-nfl.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Learning.Sharp", "declaration": "AISafetyAtlas.Learning.permInvariant_of_closedUnderPermutation", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "content_source_refs": [ "schumacher-vose-whitley-2001" ], "content_source_note": "Schumacher-Vose-Whitley's set formulation of closure under permutation, expressed as the statement that such a set's indicator is a permutation-invariant weight, so that their theorem is the indicator case of the weighted one.", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Learning.Sharp AISafetyAtlas.Examples.Learning.Sharp", "scope_delta": { "summary": "Records the bridge from the set form to the weighted form rather than transcribing the 2001 paper, whose algorithm model and problem-description-length framing are not mechanized. The set-form NFL conclusion itself follows by composing with nfl_of_permInvariant; the worked instance is Examples.Learning.nfl_over_constants.", "evidence": "docs/provenance/lean-wolpert-nfl.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Learning.nfl_adaptive_iff_permInvariant", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.nfl_adaptive_of_permInvariant", "AISafetyAtlas.Learning.permInvariant_of_nfl" ], "application": "The exact boundary of No Free Lunch on the prior axis, over the adaptive non-repeating class the sources quantify over: aggregate performance is algorithm-independent precisely when the weighting of objectives is invariant under permutations of the search domain. This is the headline statement of the row; the uniform cores below are its constant-weight case." }, { "atlas_declaration": "AISafetyAtlas.Learning.nfl_mixture_of_permInvariant", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.mixtureTrace", "AISafetyAtlas.Learning.mixtureTrace_eq_sum_mul" ], "application": "No Free Lunch for randomized search heuristics, at Droste-Jansen-Wegener's own definition of one: a distribution over the finitely many deterministic strategies. Wider than print on the mixture weight, which here may be signed and needs only equal total mass." }, { "atlas_declaration": "AISafetyAtlas.Learning.nfl_stochastic_of_permInvariant", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.stochasticTrace", "AISafetyAtlas.Learning.induced" ], "application": "The same result with the randomness drawn at each step rather than over whole strategies, which is the other picture the literature uses. surjective_induced_playChoice records that it reaches the entire mixture class." }, { "atlas_declaration": "AISafetyAtlas.Combinatorics.card_closedUnderPermutation_nonempty", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Combinatorics.closedUnderPermutationEquivSet", "AISafetyAtlas.Combinatorics.spectrum" ], "application": "Igel-Toussaint Theorem 3: the permutation-closed priors number 2^C(|X|+|Y|-1,|X|) - 1 out of 2^(|Y|^|X|) - 1, so the condition characterizing No Free Lunch is met by almost nothing. fraction_closedUnderPermutation is the printed fraction in Q." }, { "atlas_declaration": "AISafetyAtlas.Combinatorics.exists_perm_rel_not_iff", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Combinatorics.forall_rel_of_permInvariant" ], "application": "Igel-Toussaint Theorem 4: a non-trivial neighbourhood relation on the search space is not invariant under permutations of it. Stated for an arbitrary binary relation, since the proof uses neither the symmetry print assumes nor irreflexivity. Note the deliberate non-claim: the step from here to 'this family of objectives is not permutation-closed' is print's Examples 2 and 3 and is NOT formalized." }, { "atlas_declaration": "AISafetyAtlas.Learning.no_free_lunch", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.sum_performance_eq_scaled_sum", "AISafetyAtlas.Learning.aggregatePerformance_eq_scaled_sum" ], "application": "Optimization: any two injective non-adaptive m-point schedules have equal uniformly-averaged cost-sequence score. Finite-domain core of Wolpert and Macready (1997)." }, { "atlas_declaration": "AISafetyAtlas.Learning.no_free_lunch_adaptive", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.observed_of_consistent", "AISafetyAtlas.Learning.adaptive_constraint_card", "AISafetyAtlas.Learning.sum_performance_eq_scaled_sum" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.nfl_adaptive_of_permInvariant", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.PermInvariant", "AISafetyAtlas.Learning.weightedTrace", "AISafetyAtlas.Learning.observed_eq_iff" ] }, { "atlas_declaration": "AISafetyAtlas.Combinatorics.basisClass_histogram_eq_permOrbit", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Combinatorics.histogram", "AISafetyAtlas.Combinatorics.basisClass", "AISafetyAtlas.Combinatorics.permOrbit" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.exists_observed_eq", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.exists_perm_ruleVisit", "AISafetyAtlas.Learning.observed_eq_iff" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.card_observed_eq", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.no_free_lunch_adaptive_of_sharp" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.closedUnderPermutation_of_permInvariant", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Combinatorics.ClosedUnderPermutation", "AISafetyAtlas.Learning.PermInvariant" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.no_free_lunch_adaptive_of_sharp", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.nfl_adaptive_of_permInvariant" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.closedUnderPermutation_of_nfl", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.permInvariant_of_nfl", "AISafetyAtlas.Learning.closedUnderPermutation_of_permInvariant" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.ruleVisit_permRule", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.permRule" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.observed_permRule", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.permRule" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.nfl_iff_permInvariant", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.nfl_of_permInvariant", "AISafetyAtlas.Learning.permInvariant_of_nfl" ], "application": "The exact boundary of no free lunch on the prior axis: every schedule performs equally under a weighting of objectives precisely when that weighting is invariant under permutations of the search domain. Real priors over learning problems are structured, hence not permutation-symmetric, hence outside the hypothesis — which is what lets a consumer say where NFL does not bite." }, { "atlas_declaration": "AISafetyAtlas.Learning.nfl_of_permInvariant", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.PermInvariant", "AISafetyAtlas.Learning.exists_perm_comp", "AISafetyAtlas.Learning.weightedPerformance" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.permInvariant_of_nfl", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.weightedPerformance_indicator" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.permInvariant_of_closedUnderPermutation", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Combinatorics.ClosedUnderPermutation", "AISafetyAtlas.Learning.PermInvariant" ] }, { "atlas_declaration": "AISafetyAtlas.Learning.no_free_lunch_embedding_of_sharp", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Learning.nfl_of_permInvariant" ] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Lean NEW_PROOF (2026-07-19): AISafetyAtlas.Learning.no_free_lunch formalizes the finite-domain non-adaptive uniform-averaging core of Wolpert–Macready 1997 — for finite X,Y, any cost-sequence score Φ, any two injective m-point schedules have equal sum of Φ over all objectives f:X→Y (equivalently the closed form |Y|^{|X|-m}·∑_c Φ(c)). Relationship is RELATED, not EXACT/EQUIVALENT: adaptive query trees, stochastic algorithms, and time-varying objectives from the paper are not mechanized. Supervised twin is BY-020 / no_free_lunch_supervised. Distinct from SSBD PAC NFL / AFP No_Free_Lunch_ML (LAND-NFL-001) and continuous free lunches (BY-022). Provenance: docs/provenance/lean-wolpert-nfl.md. ai_interpretation_status stays HUMAN_REVIEW. CT-10 (2026-08-16, refiled here 2026-08-16): the sharp closed-under-permutation characterization lands in AISafetyAtlas.Learning.Sharp — nfl_adaptive_iff_permInvariant, graded RELATED against Igel-Toussaint 2004 (Theorem 5) and Schumacher-Vose-Whitley 2001. Aggregate performance is algorithm-independent iff the weighting of objectives is invariant under permutations of the search domain. It widens the printed theorem on the weight axis (arbitrary real weights, including signed ones, where the source takes a probability distribution) and matches it on the algorithm axis, since nfl_adaptive_of_permInvariant proves sufficiency over the printed non-repeating adaptive class; the necessity half assumes strictly less than the source (schedules only, one sample length, indicator measures) and is therefore stronger. It does not match print on sample length — Theorem 5 already quantifies over any m in {1,...,|X|}, so that is parity, not a sharpening. Stochastic algorithms and time-varying objectives remain out of scope. These records sat on BY-020 until 2026-08-16; that row is Wolpert 1996 supervised learning, and both CT-10 sources are search papers.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "isabelle-afp" ], "query_terms": [ "no free lunch", "black box optimization", "wolpert", "macready" ] }, "candidate_formalizations": [ { "repository": "https://github.com/JinZida/DSC291LT_NFL", "revision": "a24c64c5ebb56200b22b4ad6059eef7a74ab6b35", "framework": "Lean", "license": "NONE", "declaration": "NFL.CodexFull.WolpertMacready.wolpert_macready and .wolpert_macready_avg (NFL/NFL/codex_full/WolpertMacready.lean); independently duplicated as NFL.ClaudeFull.WolpertMacready.* (NFL/NFL/claude_full/WolpertMacready.lean). The statement skeleton at NFL/NFL/WolpertMacready.lean is sorry-closed.", "inspection_state": "REPRODUCED", "relationship_review": "RELATED", "notes": "UCSD DSC 291 (Learning Theory) course project, Spring 2026, formalizing three No Free Lunch theorems. Layout: NFL.lean imports a 69-line statement skeleton whose theorems are sorry-closed; the proofs live in codex_full/ and claude_full/, outputs of two AI coding agents on the same task, with the README designating codex_full/ as the deliverable. Built and audited 2026-07-28 at the repository's own pins (Lean v4.29.1, Mathlib 5e932f97). The default lake build target succeeds but warns 'declaration uses sorry' at WolpertMacready.lean lines 42 and 60; no oleans appear under codex_full/ or claude_full/, confirming both are outside the default closure. Built explicitly, both compile (1700 jobs each). Axiom audit reports [propext, Classical.choice, Quot.sound] for wolpert_macready and wolpert_macready_avg in both namespaces, with no sorryAx: both agent proofs are complete and kernel-checked. Delta against existing atlas coverage is narrow. BY-021 is already covered RELATED by the in-tree Lean AISafetyAtlas.Learning.no_free_lunch, and AISafetyAtlas.Learning.no_free_lunch_adaptive already proves the deterministic adaptive case including the full histogram (each fiber has cardinality |Y|^(|X|-m) independent of the rule). What this lead adds is stochastic algorithms: its SearchAlg.nextQuery is PMF-valued where AdaptiveRule is deterministic. It is not uniformly stronger — its performance functional is ENNReal-valued where the atlas admits signed real functionals — and it likewise omits the time-varying objectives that keep the atlas artifact at RELATED. Graded RELATED for consistency with that standard. Its other two theorems are more interesting than the Wolpert one: batch_stochastic_NFL is SSBD Thm 5.1, which the atlas holds only in Isabelle as LAND-NFL-001, though that AFP entry is strictly more general (arbitrary measure spaces, not Fintype-only), so this is a finite-domain special case rather than a Lean replacement; and exists_adversarial_target is an online adversarial NFL with no atlas analogue. The blocker is licensing: no license file, so nothing here is reusable, vendorable or promotable however good the proofs are. Version skew against atlas pins (Lean v4.33.0, Mathlib db584cd6) is secondary. Found via an incidental control query, not a systematic lead sweep of this row." } ], "tags": [ "learning-theory" ], "public": { "group": "Learning and generalization", "summary": "Same for search, but only under an exact condition: averaged over a set of objectives that cannot tell one point of the search space from another, every optimizer scores alike — and almost no set of objectives is like that.", "use": "Checking whether a \"no optimizer is better in general\" argument applies to the problem at hand at all.", "title": "No free lunch — optimization", "attribution": "Wolpert & Macready" } }, { "id": "BY-022", "name": "Free lunches in continuous spaces and coevolution", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "The cited results identify settings where standard no-free-lunch symmetry fails.", "original_source_refs": [ "survey-ref-047", "survey-ref-048" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "continuous free lunch", "coevolutionary free lunch" ] }, "tags": [ "learning-theory" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged. NEITHER SOURCE IS HELD. Auger and Teytaud is in Algorithmica 57, Springer, subscription; its HAL deposit inria-00369788 serves an HTML landing page and the file URL was not found on 2026-09-13. Wolpert and Macready, Coevolutionary free lunches, is in IEEE TEVC 9(6), subscription; a Santa Fe Institute working-paper deposit is the likely open route and was not tried. THIS ROW IS THE ONE WHERE THE TREE IS CLOSEST AND THE DISTANCE STILL MATTERS. AISafetyAtlas.Learning and AISafetyAtlas.Learning.Sharp carry the no-free-lunch theorem and its sharpness, and section 4 of docs/provenance/source-coverage-audit.md grades Igel-Toussaint and Schumacher-Vose-Whitley against them. Print here is the OTHER direction: settings in which the NFL symmetry FAILS - continuous domains in Auger and Teytaud, coevolution in Wolpert and Macready. So the candidate to adjudicate is the atlas's own NFL layer, and what has to be checked is whether its hypotheses are exactly the ones print shows can be dropped. That comparison needs the two papers and is not guessable from the existing sections." } }, { "id": "BY-023", "name": "Unidentifiability", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Observational data can be compatible with multiple latent, causal, or generative explanations.", "original_source_refs": [ "survey-ref-049", "survey-ref-050", "survey-ref-051", "survey-ref-052" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "unidentifiability", "nonidentifiability", "disentanglement impossibility" ] }, "tags": [ "preference-inference", "learning-theory" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "RETRIAGED 2026-09-13, from UNTRIAGED. TWO OF THE FOUR SOURCES ARE NOW HELD. Locatello et al. was fetched from the PMLR v97 proceedings and is pinned as locatello-etal-pmlr-2019-challenging-common-assumptions-in-the-unsupervised-learning-of-disentangled-representations.pdf, sha256 bbce8bcfc6e25f92c9e74d9cc9cbdd5039688a8cb7d572d544f9dd63e533a9ef, 11 pp.; Peters, Janzing and Scholkopf's Elements of Causal Inference is held as the MIT Press open-access deposit, sha256 52f8843f135c969f31fd93d52ac380d96ac9cdbd53c5df2214b02f80d4455598, 289 pp., not yet read. Hyvarinen and Pajunen and Bojinov and Basse are not held; routes in the private manifest of 2026-09-13. PRINT'S THEOREM 1, read 2026-09-13 from a rendered image of PMLR page 3: for d > 1, let z ~ P admit a density p(z) = prod_i p(z_i); then there exists an INFINITE FAMILY of bijections f : supp(z) -> supp(z) with every partial derivative of f_i with respect to u_j nonzero almost everywhere - so z and f(z) are completely entangled - and with the same joint law, P(z <= u) = P(f(z) <= u) for all u. SO THE OBJECT IS A MEASURE-THEORETIC NON-IDENTIFIABILITY and it is exactly statable: a reparametrisation that preserves the law and destroys coordinatewise structure. NOTHING IN THE TREE IS THIS, checked 2026-09-13: 'disentangl' occurs only in registry.yaml; AISafetyAtlas.Preference and the MAIS O31 conjecture carry identifiability language about PREFERENCES from behaviour, which is a different non-identifiability with a different carrier. A FRESH FORMALIZATION IS OWED. THE SUBSTRATE IS REACHABLE BUT NOT CHEAP: the statement needs densities on a product space, a Jacobian with an almost-everywhere condition, and a construction print defers to its appendix A, which was not read. The Gaussian case print calls 'very intuitive' is the natural first target and is narrower than print." } }, { "id": "BY-024", "name": "Physical limits on inference", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Embedded physical inference devices face limits on prediction, observation, and mutual control.", "original_source_refs": [ "survey-ref-005", "survey-ref-053", "survey-ref-054", "survey-ref-055" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Inference", "atlas_module": "AISafetyAtlas.Inference", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Inference.InferenceDevice", "AISafetyAtlas.Inference.IsProbe", "AISafetyAtlas.Inference.IsSourceProbe", "AISafetyAtlas.Inference.isSourceProbe_iff", "AISafetyAtlas.Inference.surjective_of_isProbe", "AISafetyAtlas.Inference.exists_isProbe", "AISafetyAtlas.Inference.probe", "AISafetyAtlas.Inference.WeaklyInfers", "AISafetyAtlas.Inference.weaklyInfers_iff_imageProbes", "AISafetyAtlas.Inference.weaklyInfers_iff_sourceProbes", "AISafetyAtlas.Inference.Distinguishable", "AISafetyAtlas.Inference.StronglyInfers", "AISafetyAtlas.Inference.SemiControls", "AISafetyAtlas.Inference.Controls", "AISafetyAtlas.Inference.separatingDevice", "AISafetyAtlas.Inference.separatingDevice_weaklyInfers", "AISafetyAtlas.Inference.exists_weaklyInfers_family_of_two_values_on", "AISafetyAtlas.Inference.exists_weaklyInfers_of_two_values_on", "AISafetyAtlas.Inference.not_weaklyInfers_own_concl", "AISafetyAtlas.Inference.exists_not_weaklyInfers", "AISafetyAtlas.Inference.not_infersDevice_both_of_distinguishable", "AISafetyAtlas.Inference.weaklyInfers_of_stronglyInfers", "AISafetyAtlas.Inference.stronglyInfers_trans", "AISafetyAtlas.Inference.infersDevice_of_stronglyInfers", "AISafetyAtlas.Inference.not_stronglyInfers_both", "AISafetyAtlas.Inference.not_stronglyInfers_self", "AISafetyAtlas.Inference.exists_not_stronglyInfers", "AISafetyAtlas.Inference.exists_weaklyInfers_family_of_values_attained_on", "AISafetyAtlas.Inference.weaklyInfers_of_controls", "AISafetyAtlas.Inference.semiControls_of_controls", "AISafetyAtlas.Inference.semiControls_of_controls_classical", "AISafetyAtlas.Inference.semiControls_setup_of_stronglyInfers", "AISafetyAtlas.Inference.not_controls_own_concl", "AISafetyAtlas.Inference.not_controls_both_of_distinguishable", "AISafetyAtlas.Inference.setup_partition_eq_of_semiControls_setup", "AISafetyAtlas.Inference.infersDevice_comm_of_semiControls_setup", "AISafetyAtlas.Inference.not_stronglyInfers_either_of_semiControls_setup", "AISafetyAtlas.Inference.not_controls_other_setup_of_semiControls_setup", "AISafetyAtlas.Inference.weaklyInfers_iff_three_setups", "AISafetyAtlas.Inference.not_weaklyInfers_of_at_most_two_setups", "AISafetyAtlas.Inference.exists_stronglyInfers_of_large_fibres", "AISafetyAtlas.Inference.inferenceComplexity", "AISafetyAtlas.Inference.inferenceComplexityTotal", "AISafetyAtlas.Inference.inferenceComplexity_le_of_stronglyInfers", "AISafetyAtlas.Inference.Mimics", "AISafetyAtlas.Inference.Copies", "AISafetyAtlas.Inference.Mimics.trans", "AISafetyAtlas.Inference.Copies.symm", "AISafetyAtlas.Inference.Copies.trans", "AISafetyAtlas.Inference.FullReality", "AISafetyAtlas.Inference.FullReality.SourceStipulations", "AISafetyAtlas.Inference.FullReality.reducedForm", "AISafetyAtlas.Inference.AdmissibleTuples", "AISafetyAtlas.Inference.IsReducedForm", "AISafetyAtlas.Inference.lemma1_reducedForm_iff_admissible", "AISafetyAtlas.Inference.exists_pairwise_distinguishable_weak_cycle", "AISafetyAtlas.Inference.not_mutually_distinguishable_weak_cycle", "AISafetyAtlas.Inference.not_strong_inference_cycle", "AISafetyAtlas.Inference.unique_strong_root", "AISafetyAtlas.Inference.exists_copies_distinguishable_weak", "AISafetyAtlas.Inference.exists_copies_stronglyInfers_infinite", "AISafetyAtlas.Inference.copies_stronglyInfers_not_finite", "AISafetyAtlas.Inference.FinPMF", "AISafetyAtlas.Inference.positiveMassSetups", "AISafetyAtlas.Inference.inferenceAccuracy", "AISafetyAtlas.Inference.inferenceAccuracy_eq_of_two_setups", "AISafetyAtlas.Inference.miDistinguishability", "AISafetyAtlas.Inference.miDistinguishability_mem_unit_interval", "AISafetyAtlas.Inference.countingDistinguishability", "AISafetyAtlas.Inference.Prop6Quadruple", "AISafetyAtlas.Inference.StatisticallyIndependent", "AISafetyAtlas.Inference.Prop6Law", "AISafetyAtlas.Inference.prop6Law_of_independent", "AISafetyAtlas.Inference.prop6_product_eq", "AISafetyAtlas.Inference.prop6_half", "AISafetyAtlas.Inference.prop6Expr_half_maximizer", "AISafetyAtlas.Inference.SelfAwareDevice", "AISafetyAtlas.Inference.Intelligible", "AISafetyAtlas.Inference.Infallible", "AISafetyAtlas.Inference.Corrects", "AISafetyAtlas.Inference.weaklyInfers_of_infallible_semiControls_question", "AISafetyAtlas.Inference.stronglyInfers_of_infallible_semiControls_question_setup", "AISafetyAtlas.Inference.thm7_card", "AISafetyAtlas.Inference.thm7_ii", "AISafetyAtlas.Inference.probeImage_eq_of_card_eq", "AISafetyAtlas.Inference.thm7_ii_chain", "AISafetyAtlas.Inference.not_mutually_deviceIntelligible_of_finite", "AISafetyAtlas.Inference.not_both_concl_intelligible", "AISafetyAtlas.Inference.exists_not_corrects", "AISafetyAtlas.Inference.setupMass", "AISafetyAtlas.Inference.cellAgreeProb", "AISafetyAtlas.Inference.prop6QuadrupleOf", "AISafetyAtlas.Inference.eq_of_apply_eq_of_card_eq", "AISafetyAtlas.Inference.mutualInfo_nonneg", "AISafetyAtlas.Inference.mutualInfo_le_add", "AISafetyAtlas.Inference.gibbs_cell", "AISafetyAtlas.Inference.sum_pushOnImage", "AISafetyAtlas.Inference.pushOnImage_marginal" ], "build_command": "lake build AISafetyAtlas.Inference AISafetyAtlas.Examples.Inference.Device", "scope_delta": { "summary": "Wolpert 2008 (survey-ref-005) only, read from arXiv:0708.1362v2 and cross-checked statement by statement against the published Physica D 237(9):1257-1281 article. All 46 tracked statements (the numbered inventory plus Theorem 2's unnumbered third sentence) have Lean declarations and per-item statuses in docs/provenance/wolpert-inference-devices.md: 42 SOURCE-EXACT, 1 SPECIALIZED, 2 REPAIRED, 0 ASSUMED-STEP, 1 REFUTED, 0 unmechanized. RELATED is retained because the row cites three other untranscribed works and because the transcription includes repairs, specializations and one machine-checked refutation. Proposition 6 no longer assumes any step of its source proof: mutualInfo_eq_zero_iff proves the equality case of Gibbs inequality, so statisticallyIndependent_of_miDistinguishability_eq_one derives the sentence the paper asserts without proof, and prop6_half_of_miDistinguishability_eq_one proves the 1/4 bound from the printed mutual-information-distinguishability-one premise. It stays SPECIALIZED for finiteness and for two side conditions the printed statement leaves implicit: positive mass on the four setup fibres, and nonvanishing total setup entropy. Corollary 1(ii) is false as printed and is refuted; its proof requires the proper subset W that the statement omits. Proposition 7 is repaired because the paper's displayed witness is not a Definition 12 device; the Atlas witness satisfies the relaxed Lean device model but not the paper's separate global convention that each function in a reality have at least two image values. Corollary 5 drops a source hypothesis made vacuous by Theorem 1. The finite specializations are Definition 6, Theorem 4, Proposition 4 and Theorem 7(i); other representational scope differences are recorded item by item. FullReality now preserves the source's nonempty-family premise. Both halves of Theorem 6 have executable witnesses. Definitions 9-11 use a finite probability mass function; section 8 remains a local finite/discrete formalization with no PFR dependency. All encoding readings, source defects, omitted prose definitions and boundary conventions are recorded in docs/provenance/wolpert-2008-source-clashes.md. The deterministic impossibility results remain fully general over arbitrary U. Nothing in the Lean layer establishes the paper's physical-worldline interpretation of U.", "evidence": "docs/provenance/wolpert-inference-devices.md" } } ], "lean_artifact": null, "public": { "group": "What an observer can recover", "title": "Physical limits on inference", "summary": "No device inside a system can answer every question about that system, no two distinguishable devices can each infer the other, and no two devices can each emulate the other.", "use": "Holds for any observation, prediction or recollection device, whatever the physical laws; control inherits every limit.", "attribution": "Wolpert" }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Wolpert 2008 is graded RELATED. The published article's 46 tracked statements are covered at per-item status 42 SOURCE-EXACT, 1 SPECIALIZED, 2 REPAIRED, 0 ASSUMED-STEP and 1 REFUTED; the complete map is docs/provenance/wolpert-inference-devices.md. Proposition 6 is proved from its printed premise, the equality case of Gibbs inequality having closed the step the paper asserts without derivation. Corollary 1(ii) is machine-refuted as written. Proposition 7 is repaired under an explicitly relaxed global image convention. Corollary 3(ii) is both directions under mutual semi-control, not Theorem 3. Section 8 is finite/discrete and adds no PFR dependency. Weak inference and Knowledge.Knowable are incomparable, with both countermodels exhibited. The other three cited works remain untranscribed, and nothing here makes U a set of physical worldlines. Definition 9 is now SOURCE-EXACT: inferenceAccuracySupOn uses the printed supremum over positive-mass setup fibres on an arbitrary setup range, and no setup measurability hypothesis is retained. Definition 10 likewise uses tsum over an unrestricted value type subject only to the finite-entropy conditions its printed ratio needs. Definition 11 remains the sole numbered SPECIALIZED item because its printed ratio of cardinalities is undefined as infinity divided by infinity on countably infinite setup ranges; choosing a totalization would add mathematics not supplied by the source.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "physical limits of inference", "inference device", "inference devices", "computational capabilities of physical systems", "wolpert" ] }, "tags": [ "information-theory" ] }, { "id": "BY-025", "name": "Uncontainability", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Perfect containment of a superintelligent program is undecidable in the formal model used by the source.", "original_source_refs": [ "survey-ref-056" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Verification.Containment", "declaration": "AISafetyAtlas.Verification.Containment.harming_undecidable", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Verification.Containment AISafetyAtlas.Examples.Verification.Containment", "scope_delta": { "summary": "SAME as print's Theorem 1 in the claim - no total computable decider for the harm predicate - and the proof is print's, holding the program fixed at HaltHarm and letting the data range over encoded machines, which is the argument on journal page 71. WIDER in the carrier: harms is an ARBITRARY relation between programs and data, so nothing in the Lean means 'harms humans'; print's HarmHumans() is likewise an opaque operation of the language of R, so this widening costs nothing and the safety reading is the bridge rather than the theorem. WIDER in the hypothesis: HarmReduction asks only for a distinguished program, a computable packaging of a machine as data, and print's biconditional, where print's Assumption 2 also asserts universal simulation; the proof never simulates, and Examples.Verification.Containment supplies the simulating witness anyway, built from Nat.Partrec.Code.exists_code applied to the universal evaluation function, so the antecedent is inhabited by a real construction and not by an identity. THE AXIS THAT WAS NARROWER CLOSED 2026-09-20: print quantifies over every machine T and every input I, and harming_undecidable fixes sourceInput and varies the machine, because that is the shape of Mathlib's ComputablePred halting statement. Print's quantifier is now here too - HarmReductionPair and harming_undecidable_pair - and HarmReductionPair.toHarmReduction is the reduction between the two, which is the equivalence this note used to assert in prose. The fixed-input form is the STRONGER theorem, since its hypothesis asks the composite to work at one input only. It cost one lemma Mathlib does not state, halting_problem_pair, the halting problem at both arguments. PRINT'S ASSUMPTION 2 IS ALSO A STATEMENT NOW, as Assumption2, in print's three clauses, with harming_undecidable_of_assumption2 deriving Theorem 1 from it and nothing else; it is deliberately stronger than the proof consumes, its simulation clause being an equality of partial functions because a machine agreeing only on domains would not be simulating. Examples.Verification.Containment.exists_assumption2 inhabits it with a composite that unpairs its input into a machine and an input for it. Verification.Robot still narrows in the way this module used to. NOT CARRIED AT ALL: print's Corollary 3, this row's own informal claim, for the reason in the row note - print derives it in prose, gives no reduction, and never defines the containment problem, its page-69 glossary defining only when a MACHINE is containable. Nor is print's own containment notion the field's: page 68's boxing sense, where the system may act to escape an isolation boundary, is surveyed and then dropped, and nothing here touches it.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Verification.Containment.harming_undecidable", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Computability.halting_problem" ], "application": "Containment: no total computable procedure decides, for every program and every input, whether that program on that input has a designated effect, once the effect can be appended to a simulated computation. This is Alfonseca et al. Theorem 1 (JAIR 70:65-76, journal page 71) reduced to the halting problem. The construction certificate is witnessed in `Examples.Verification.Containment` by a program extracted from the universal evaluation function.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-harming-undecidable.md" } } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Carries atlas Lean since 2026-09-13, graded RELATED. WHAT THE ATLAS HAS AND WHAT THIS ROW CLAIMS ARE NOT THE SAME STATEMENT, and that is the whole scope note. This row's informal claim is about CONTAINMENT, which is print's Corollary 3. AISafetyAtlas.Verification.Containment.harming_undecidable is print's THEOREM 1, the harming problem. Print proves Theorem 1 by substitution into Algorithm 3 and then derives Corollary 3 in running prose from Assumption 2 - 'the impossibility of deciding whether a superintelligent machine R will harm humans in the presence of a scenario D leads to the conclusion that the containment problem cannot be computed' - without a reduction. AND THERE ARE TWO SENSES OF CONTAINMENT IN THE PAPER, which is why this row needs the distinction spelled out. Journal page 68 surveys the sense the rest of the field uses - isolation, restricted channels, a system that may act to get out, after Lampson's 1973 confinement problem, with Yudkowsky's social-engineering result and Yampolskiy's binary-question channel cited by name. Journal page 69's glossary then defines its own: 'A superintelligent machine is containable if there is a control strategy that prevents its acting on the external world when there is a reason to predict that R(D) will harm humans, and allows it otherwise.' That is PREDICT-THEN-GATE, a property of a machine rather than a decision problem, and the escape half is dropped between the two pages. So the containment problem whose incomputability Corollary 3 asserts is never defined at all, there is no argument to transcribe, and none is claimed; stating the complement of the harm predicate and calling it containment would be inventing print's missing step. A READER WHO ARRIVES WITH THE FIELD'S DEFINITION WILL OVER-READ THIS ROW: neither print's Theorem 1 nor the atlas declaration says anything about recognizing a system trying to breach an isolation boundary. The atlas therefore holds the load-bearing INPUT to this row's claim and not the claim. WHAT WAS FOUND BEFORE THE BUILD, and why the row was UNTRIAGED until 2026-09-13: RETRIAGED 2026-09-13, from UNTRIAGED. THE SOURCE IS NOW HELD IN ITS PUBLISHED FORM AND READ. Alfonseca, Cebrian, Fernandez Anta, Coviello, Abeliuk and Rahwan was fetched from the JAIR article server and is pinned as alfonseca-cebrian-fernandez-anta-coviello-abeliuk-rahwan-jair-2021-superintelligence-cannot-be-contained.pdf, sha256 ee8cfd460b9074e085142f84ddbfee03606a55954f7e5125e9f7d79c39f0b1a4. The running head on the face reads 'Journal of Artificial Intelligence Research 70 (2021) 65-76, Submitted 06/2020; published 01/2021', so this is the published text and not a preprint. Journal page 71 read from a rendered page image. WHAT PRINT PROVES: 'Theorem 1. The harming problem is undecidable', by a reduction from halting. Print's Algorithm 3 is HaltHarm(T, I) = execute T(I); execute HarmHumans(); then a decider for Harm(R, D) at R = HaltHarm and D = (T, I) returns TRUE exactly when T(I) halts. 'Corollary 3. The containment problem is incomputable' follows under print's Assumption 2, that the program simulates a universal machine and its language contains an explicit HarmHumans operation taking finite time. SO THIS IS A HALTING REDUCTION AND THE ATLAS HAS THE SOURCE OF IT: AISafetyAtlas.Computability.halting_problem, halting_re and nonhalting_not_re, and Rice's theorem at BY-012. What the atlas does NOT have, checked 2026-09-13, is any predicate standing for 'harms humans' or any containment problem - neither word occurs in AISafetyAtlas/. A FRESH FORMALIZATION IS OWED AND IT IS SMALL. The whole content is: given an oracle deciding a predicate P of (program, input) such that the composite program 'run T(I) then do X' satisfies P exactly when T(I) halts, halting is decidable. That is one lemma over the atlas's existing halting layer, and print's Assumption 2 is exactly the hypothesis that makes the composite constructible. The honest scope note is that print's HarmHumans is an uninterpreted operation, so the formal statement is about an ARBITRARY effect predicate and the safety reading is the bridge, not the theorem.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "uncontainability", "superintelligence cannot be contained", "containment problem" ] }, "tags": [ "computability", "verification" ], "public": { "group": "Limits of verification", "summary": "No procedure decides exactly, for every program and input, whether it will perform an effect that can be appended to an arbitrary computation -- the kind of effect you are trying to prevent.", "use": "Deciding what a safety check before deployment can promise.", "title": "The harming problem", "attribution": "Alfonseca et al." } }, { "id": "BY-026", "name": "Uninterruptibility", "paper_reference": "Table 1", "survey_proof_assessment": "NOT_PROVEN_IN_SURVEY", "informal_claim": "Available interruption proposals guarantee safe interruptibility only under restricted assumptions.", "original_source_refs": [ "survey-ref-057", "survey-ref-058", "survey-ref-059", "survey-ref-060", "survey-ref-061" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "safe interruptibility", "uninterruptibility", "off-switch", "interruptibility", "safely interruptible", "corrigibility" ] }, "tags": [ "agent-incentives" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged because two of the five sources could not be obtained. THREE ARE NOW HELD, all fetched 2026-09-13 and pinned in the private manifest of 2026-09-13: Carey, Incorrigibility in the CIRL Framework, arXiv:1709.06275v2, sha256 696cb676dd2a0525ba4a4a637e937c6dabef984d85a9805c76241d1f191c25c2, carrying Definition 1 and Theorem 1; Hadfield-Menell, Dragan, Abbeel and Russell, The Off-Switch Game, arXiv:1611.08219v3, sha256 2e8131110d1921e8de0a096ea3f39a2012be684532ba58db0105186e0eae0f7d, carrying Theorems 1 and 2 and Corollary 1; and Orseau and Armstrong, Safely Interruptible Agents, from the UAI 2016 proceedings server, sha256 bfb4209e0c3f517dea29003b6ccb69dfb10a28897949f2912c4c7e79b3d0607e, carrying Definitions 1 to 19, Lemmas 3 to 24, Theorems 7, 8, 14, 15, 17 and 18 and Proposition 11. Statement inventories only; no statement of any of the three has been read against the tree, so no grade is claimed. TWO ARE NOT HELD: Wangberg, Boors, Catt, Everitt and Hutter, A Game-Theoretic Analysis of the Off-Switch Game, AGI 2017, Springer LNAI 10414 - note the atlas holds the AGI 2016 volume, which is the wrong year; and El Mhamdi, Guerraoui, Hendrikx and Maurer, Dynamic safe interruptibility, NIPS 2017, whose proceedings server is the route. A FETCH DEFECT WORTH RECORDING: two arXiv identifiers were guessed from memory for the last two and both returned a DIFFERENT PAPER - 1703.02010 is a dynamical-systems paper on shadowable chain transitive sets, and 1708.02786 is a statistics paper on structural stability in factor models. Both files were deleted rather than kept. Recorded so the identifiers are not tried again. WHAT THE TREE HAS, CHECKED 2026-09-13: nothing. 'interrupt' occurs in AISafetyAtlas/Upstream/Debate/Protocol.lean in an unrelated sense and in this ledger; 'off-switch' occurs in the ledger alone; there is no shutdown or interruptibility object anywhere in AISafetyAtlas/. The nearest real neighbour is the wireheading and self-modification cluster, which is about an agent changing its own utility rather than about an agent's incentive to permit a shutdown." } }, { "id": "BY-027", "name": "Löb's theorem (unverifiability)", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Löb's theorem: if a sufficiently strong formal system T proves Prov_T(σ) → σ, then T already proves σ (under the source's arithmetic hypotheses).", "original_source_refs": [ "survey-ref-062" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/FormalizedFormalLogic/Foundation", "version": "30a16ffa93d79d73ab4d02427fa00f50e039bf29", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; Foundation@30a16ffa93d79d73ab4d02427fa00f50e039bf29", "build_command": "lake build AISafetyAtlas", "module": "Foundation.FirstOrder.Incompleteness.Löb", "declaration": "LO.FirstOrder.Arithmetic.löb_theorem", "relationship": "EQUIVALENT" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Logic.loeb", "type": "WRAPPER", "source_declarations": [ "LO.FirstOrder.Arithmetic.löb_theorem" ] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Löb's theorem (if T proves Prov_T(σ)→σ then T proves σ) is reproduced from FormalizedFormalLogic/Foundation and exposed via AISafetyAtlas.Logic.loeb. Classified EQUIVALENT to the survey row named \"Löb's theorem (unverifiability)\": the formalization matches the Löb schema itself. The informal AI-safety reading \"cannot prove own soundness\" is a standard consequence under the theorem's hypotheses (Δ₁ theory interpreting IΣ₁), recorded as interpretation context only and not as a separate coverage claim. AFP Loebs_Theorem and HOL Light GL remain discovery provenance only. (Re-triaged 2026-07-19 per adversarial review R6-5.)", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "isabelle-afp", "hol-light" ], "query_terms": [ "loebs theorem", "loeb theorem", "loeb formula", "provability logic" ] }, "tags": [ "provability-logic", "verification" ], "public": { "group": "Limits of self-knowledge and reflection", "summary": "Saying “if I prove it, it is true” already proves it.", "use": "Why an agent cannot simply trust its own reasoning.", "title": "Löb's theorem", "attribution": "Löb" } }, { "id": "BY-028", "name": "Unpredictability of superhuman AI", "paper_reference": "Table 1", "survey_proof_assessment": "PROOF_SKETCH", "informal_claim": "A strictly less capable predictor cannot accurately predict every action of a more capable agent under the source's definitions.", "original_source_refs": [ "survey-ref-063", "survey-ref-064" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "superhuman unpredictability", "unpredictability of ai" ] }, "tags": [ "multi-agent" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged. NEITHER SOURCE IS HELD. Vinge's VISION-21 piece exists at the CMU mirror as HTML and not as a document this repository can pin by sha256 in the usual way; Yampolskiy, Unpredictability of AI, JAIC 7(1), is World Scientific and subscription. Both checked 2026-09-13. A NEIGHBOURING TEXT BY THE SAME AUTHOR IS NOW HELD AND IS NOT THIS ONE: yampolskiy-arxiv-v1-2020-on-controllability-of-ai.pdf, fetched 2026-09-13, which backs BY-040. It summarises the unpredictability argument and cites the JAIC paper rather than restating it, so it cannot stand in for the source here. WHAT THE TREE HAS THAT IS ADJACENT AND DISTINCT: BY-024's Wolpert inference-device impossibility, and the Lawvere and Brandenburger-Keisler diagonal layer. The row's claim - a strictly less capable predictor cannot predict every action of a more capable agent - is a CAPABILITY-ORDERED statement, and none of the existing diagonal results is indexed by capability. Whether print's argument is a diagonal one at all is the question a reading would settle." } }, { "id": "BY-029", "name": "Unexplainability", "paper_reference": "Table 1", "survey_proof_assessment": "NOT_PROVEN_IN_SURVEY", "informal_claim": "The cited argument links complete AI explanation to proof, truth, and model-access assumptions.", "original_source_refs": [ "survey-ref-065" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "unexplainability", "explanation impossibility" ] }, "tags": [ "interpretability" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "RETRIAGED 2026-09-13, from UNTRIAGED, with one caveat stated up front. THE PREPRINT IS HELD AND READ; THE PUBLISHED TEXT IS NOT. yampolskiy-arxiv-v1-2019-unexplainability-and-incomprehensibility-of-ai.pdf, sha256 17d210ef6ed75966ac63b84b960bcf611276742b9112fb22d2770f9c7bd2c319, 14 pp., dated June 20 2019 on its own face, fetched 2026-09-13. survey-ref-065 cites the Journal of Artificial Intelligence and Consciousness version, which is behind a World Scientific subscription and has NOT been compared with this one. Everything below is about the preprint. THE DECIDING FACT: the preprint states no theorem, lemma, proposition or corollary of its own. A full-text scan for those keywords returns section prose, citations and a bibliography. Its own words are that it presents 'impossibility results (Unexplainability and Incomprehensibility)' and it supports them by citing other people's theorems - Charlesworth's is named explicitly as 'Charlesworth proved his', and a list of known impossibility results is cited rather than restated. SO THERE IS NOTHING HERE TO ADJUDICATE A CANDIDATE AGAINST, and the six-corpus sweep's zero is not a gap in the sweep. A FRESH FORMALIZATION IS OWED, and what it would have to state is NOT in this paper: it is in the sources this paper cites, chiefly Charlesworth, which is ACM TOCL and not held. That is why BY-030 remains UNTRIAGED while this row does not - BY-030's second source IS Charlesworth. VOCABULARY NOTE, so the verdict is not read as more than it is: the statability vocabulary has no value for a source that states no theorem of its own. TRIAGED_DISTINCT is the nearest, and its gloss's first clause - candidates were examined and found distinct - is VACUOUS here rather than satisfied, because the sweep examined nothing: there is no printed statement for a candidate to be distinct from. Its second clause, that a fresh formalization is owed, does hold and is what this note above says. UNTRIAGED is wrong, not weaker: its gloss is that nobody has compared the source to the tree, and that comparison is now done. Whether the vocabulary gains a value for this case is a maintainer decision and is not taken here." } }, { "id": "BY-030", "name": "Incomprehensibility", "paper_reference": "Table 1", "survey_proof_assessment": "MIXED", "informal_claim": "Some correct computations or explanations cannot be comprehended by a bounded verifier under the cited formalizations and assumptions.", "original_source_refs": [ "survey-ref-065", "survey-ref-066" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "incomprehensibility", "software comprehension" ] }, "tags": [ "interpretability" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged because the source that carries the mathematics is not held. ONE OF THE TWO IS HELD AS A PREPRINT: yampolskiy-arxiv-v1-2019-unexplainability-and-incomprehensibility-of-ai.pdf, read 2026-09-13, which states no theorem of its own - see BY-029's note for the scan that establishes that, and for the caveat that the published JAIC text has not been compared. THE ONE THAT MATTERS IS NOT HELD: Charlesworth, Comprehending software correctness implies comprehending an intelligence-related limitation, ACM TOCL 7(3), subscription. Yampolskiy's own text names Charlesworth as having PROVED the result, so this row's mathematical content is there and nowhere else in its source list. THE TREE, CHECKED 2026-09-13: AISafetyAtlas.Verification carries verifier-shaped material and Rice's theorem is at BY-012, so a comprehension-versus-correctness statement would land next to existing computability work rather than in empty space. Whether it reduces to Rice is exactly what reading Charlesworth would settle, and is not assumed here." } }, { "id": "BY-031", "name": "k-incomprehensibility", "paper_reference": "Table 1", "survey_proof_assessment": "DEFINITIONS_ONLY", "informal_claim": "The cited source defines intelligence and comprehension limits using an intensional Kolmogorov-complexity variant.", "original_source_refs": [ "survey-ref-067" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "k-incomprehensibility", "kolmogorov intelligence" ] }, "tags": [ "interpretability", "algorithmic-information" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged. THE SOURCE IS NOT HELD and the ledger gives it NO LOCATOR - survey-ref-067, Hernandez-Orallo, A formal definition of intelligence based on an intensional variant of Kolmogorov complexity, has an empty locator field. The author's institutional deposit is the likely route and was not tried; recorded in the private manifest of 2026-09-13. WHAT THE TREE HAS, AND IT IS SUBSTANTIAL: AISafetyAtlas/Upstream/KolmogorovMathlib carries Kolmogorov complexity including its uncomputability, and AISafetyAtlas.Inference.Complexity has a halting layer over prefix-free machines. So unlike most rows in this group the substrate for an intensional-complexity definition is already here. THAT IS A REASON TO READ THE SOURCE, NOT A REASON TO GRADE IT: nothing is known about whether print's intensional variant is the one the vendored development defines, and the row says nothing about it until somebody looks." } }, { "id": "BY-032", "name": "Unverifiability", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "A universal verifier for arbitrary computations cannot satisfy the cited completeness and correctness requirements.", "original_source_refs": [ "survey-ref-068" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "isabelle-afp" ], "query_terms": [ "unverifiability", "verifier theory" ] }, "tags": [ "verification" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "RETRIAGED 2026-09-13, from UNTRIAGED, with the same version caveat as BY-029. THE PREPRINT IS HELD AND READ; THE PUBLISHED TEXT IS NOT. The file is yampolskiy-arxiv-v1-2016-verifier-theory-and-unverifiability.pdf, sha256 d07efd1c4fd3ece1e91a0c377cb100db35fe7fd4ca158e7da03e839ea54215c7, 13 pp. ITS OWN TITLE IS 'Verifier Theory and Unverifiability', which is NOT the Physica Scripta title survey-ref-068 cites; the published version is behind a subscription and has not been compared. THE DECIDING FACT: the preprint states no numbered result and no prose theorem. A full-text scan for theorem, lemma, proposition, corollary and claim returns the word 'proof' in its ordinary sense and a bibliography entry. Print's own abstract says what it is: a proposal that 'the mathematical community may be interested in studying different types of proof verifiers (people, programs, oracles, communities, superintelligences) as mathematical objects' - a research programme, not a result. SO THERE IS NOTHING TO ADJUDICATE A CANDIDATE AGAINST. This row's sweep flagged isabelle-afp as a candidate corpus and produced no candidate; that remains the only corpus note on the row. A FRESH FORMALIZATION IS OWED AND ITS SHAPE IS NOT IN PRINT. The atlas already has the pieces a verifier-theoretic impossibility would use - AISafetyAtlas.Verification, Rice's theorem at BY-012, and the Computability halting layer - and what is missing is a PRINTED statement to state, which this source does not supply. Anyone building here is building an atlas-original result and must say so rather than cite this paper for a theorem. VOCABULARY NOTE, so the verdict is not read as more than it is: the statability vocabulary has no value for a source that states no theorem of its own. TRIAGED_DISTINCT is the nearest, and its gloss's first clause - candidates were examined and found distinct - is VACUOUS here rather than satisfied, because the sweep examined nothing: there is no printed statement for a candidate to be distinct from. Its second clause, that a fresh formalization is owed, does hold and is what this note above says. UNTRIAGED is wrong, not weaker: its gloss is that nobody has compared the source to the tree, and that comparison is now done. Whether the vocabulary gains a value for this case is a maintainer decision and is not taken here." } }, { "id": "BY-033", "name": "Unverifiability of robot ethics", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Theorem 1 (van Leeuwen & Wiedermann 2021): for any non-trivial robot property P there is no algorithmic procedure that tells, for an arbitrary robot with potentially unbounded memory, whether its actions always satisfy P. Corollary 1: same conclusion restricted to robots programmed by structured programs. (Paper Def. 3 defines non-triviality via a switchable base program pair; ethics/legality are motivating instances, not the only P.)", "original_source_refs": [ "survey-ref-069" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Verification.Robot", "declaration": "AISafetyAtlas.Verification.Robot.action_safety_unverifiable", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Verification.Robot AISafetyAtlas.Examples.Verification.Robot", "scope_delta": { "summary": "Lean packages the switching construction as SwitchingConstruction and machine-checks only the reduction half; the paper's SPA syntax, its Definition 3 non-triviality construction, and the paced Turing-machine simulation are not mechanized. Bridge is REVIEWED while the formalization stays RELATED.", "evidence": "docs/guide/robot-verification-model.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Verification.Robot.action_safety_unverifiable", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Computability.halting_problem" ], "application": "Robot ethics: no total online verifier decides whether a reactive program's actions are always acceptable, given a switching certificate. Conditional core of van Leeuwen and Wiedermann, Theorem 1 (UU-PCS-2021-02), reduced to the halting problem. The certificate is witnessed in `Examples.Verification.Robot`.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-07-19", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/ct3-robot-review-package.md" } } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "REVIEWED", "notes": "Extracted paper statements (Theorem 1, Corollary 1, Defs 2–3) and a step-by-step proof comparison are in docs/guide/robot-verification-model.md (source: UU-PCS-2021-02 §3.2). The paper proof is longer than Lean's because it constructs τ_x (Fig. 1) from Def. 3 non-triviality inside Sense–Plan–Act (paced TM simulation, hyper-command validity, unbounded memory), then reduces diagonal K to the observer problem; Lean packages τ as SwitchingConstruction and only machine-checks the reduction half (fixed-input halting via Mathlib, equally undecidable). A concrete paced-computation example witnesses the assumption for Mathlib codes. Relationship is RELATED, not EXACT/EQUIVALENT: SPA syntax, Def. 3 derivation, probability, physical dynamics, and ethical/legal interpretation are not mechanized. Does not cover bounded/finite-state systems; does not exclude sound incomplete verification. Paper-model correspondence remains HUMAN_REVIEW. CT-3 2026-07-19: maintainer REVIEWED (statement + scoped interpretation); formalization relationship remains RELATED (SwitchingConstruction assumed; not paper EXACT); evidence docs/interpretation-reviews/ct3-robot-review-package.md.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "robot ethics verification", "ethical robot" ] }, "interpretation_review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-07-19", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/ct3-robot-review-package.md" }, "tags": [ "verification", "ethics" ], "public": { "group": "Limits of verification", "summary": "No monitor can guarantee a robot with unlimited memory always behaves.", "use": "Watching systems that act in the world.", "title": "Online robot verification", "attribution": "van Leeuwen & Wiedermann" } }, { "id": "BY-034", "name": "Intractability of bottom-up ethics", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Computational complexity obstructs exhaustive bottom-up evaluation of ethically relevant action consequences.", "original_source_refs": [ "survey-ref-070", "survey-ref-071" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "bottom-up ethics", "machine ethics complexity" ] }, "tags": [ "computational-complexity", "ethics" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged. ONE OF THE TWO IS NOW HELD: Brundage, Limitations and Risks of Machine Ethics, fetched 2026-09-13 from the author's own deposit and pinned as brundage-jetai-2014-limitations-and-risks-of-machine-ethics.pdf, sha256 b0914881b7dfa613b3742a6a46b9d7e3dd8c9c6bc521142f23e9c33bc454499e, 20 pp. A full-text scan finds NO numbered statement of any kind; it is a critical essay, and the complexity claim this row is about is not proved in it. THE SOURCE THAT CARRIES THE CLAIM IS NOT HELD: Reynolds, On the Computational Complexity of Action Evaluations, whose route is the MIT Media Lab publications server. Until that is read the row cannot move, because the row's claim - that complexity obstructs exhaustive bottom-up evaluation of action consequences - is a complexity statement and Brundage does not make one. THE TREE, CHECKED 2026-09-13: no ethics-evaluation object anywhere; the nearest complexity material is in AISafetyAtlas.Inference.Complexity and in the Verification cluster, neither of which is about enumerating consequences of actions." } }, { "id": "BY-035", "name": "No-flattening theorems for deep learning", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "Some functions efficiently represented by deep networks require inefficiently large shallow representations.", "original_source_refs": [ "survey-ref-072" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "A2 (2026-07-19): AFP Deep_Learning (Bentkamp; Cohen et al. network-capacity / deep-vs-shallow theorems) is RELATED landscape LAND-DL-001 — matches the informal no-flattening claim but is not EXACT formalization of the Lin–Tegmark–Rolnick 2017 survey citation (survey-ref-072). See docs/provenance/a2-deep-learning-by035-triage.md.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "no-flattening", "deep shallow network" ] }, "tags": [ "computational-complexity", "learning-theory" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "A2 (2026-07-19) found AFP Deep_Learning RELATED and recorded it as LAND-DL-001; it formalizes Cohen et al.-style network-capacity theorems rather than the Lin-Tegmark-Rolnick 2017 citation this row carries. Evidence: docs/provenance/a2-deep-learning-by035-triage.md." } }, { "id": "BY-036", "name": "Efficiency of computing Boolean functions for multilayered perceptrons", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "The cited report constrains which Boolean functions multilayer perceptrons compute efficiently.", "original_source_refs": [ "survey-ref-073" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "multilayer perceptron", "boolean functions neural" ] }, "tags": [ "computational-complexity", "learning-theory" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged. THE SOURCE IS NOT HELD. Calude, Heidari and Sifakis, What Neural Networks Are (Not) Good For?, is a CDMTCS research report at Auckland; the ledger records no report number, and a guessed URL in that series returned 404 on 2026-09-13. The route is the CDMTCS report index, which needs the number. Recorded in the private manifest of 2026-09-13. WHAT THE TREE HAS THAT WOULD BE THE COMPARATOR: Boolean-function material exists in AISafetyAtlas.Compositional.Rectangularity and in two Examples modules, and the no-free-lunch layer in AISafetyAtlas.Learning is about which functions a search can find rather than which a network can compute efficiently. Neither is a circuit-complexity statement about multilayer perceptrons, so nothing here is a candidate; but the row cannot be graded distinct from a candidate that nobody has read." } }, { "id": "BY-037", "name": "Goodhart's law (Strathern)", "paper_reference": "Table 1", "survey_proof_assessment": "NOT_PROVEN_IN_SURVEY", "informal_claim": "When a measure becomes a target, optimization pressure can destroy its value as a measure.", "original_source_refs": [ "survey-ref-074", "survey-ref-075" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "goodhart", "measure becomes a target", "regressional goodhart", "winner's curse", "regression to the mean" ] }, "candidate_formalizations": [ { "repository": "https://github.com/audieleon/goodhart", "revision": "29128f3f9bcafb30d019682b63c1b582bcadf7b9", "framework": "Lean", "license": "Apache-2.0", "declaration": "Skalse.skalse_theorem1 (proofs/GoodhartProofs/Skalse/Theorem1.lean); also skalse_theorem3, skalse_corollary3, and a potential-shaping layer (ng_vstar_shaped, ng_qstar_shaped, ng_shaping_preserves_optimal, ng_necessity_lemma3) over a FiniteMDP/Bellman development", "inspection_state": "REPRODUCED", "relationship_review": "RELATED", "notes": "Strongest of four Lean leads found by manual GitHub search on 2026-07-28; outside the six pinned corpora. Formalizes Skalse et al., 'Defining and Characterizing Reward Hacking' (NeurIPS 2022), with policies as occupancy vectors and value as an inner product (value R F = sum_i R i * F i); definitions checked non-degenerate at source level. Also carries Ng, Harada & Russell (1999) potential-based shaping over a FiniteMDP/Bellman development. Its ng_necessity_lemma3 proves an algebraic action-dependent ordering reversal, not the full Ng necessity theorem; an earlier note here overstated that. Independently built 2026-07-28 in an isolated elan 4.2.3 toolchain at the repository's own pins (Lean v4.30.0-rc2, Mathlib 9268b22206b0425419498769f780a91dee03bcf3) with 'lake exe cache get' then 'lake build': completed successfully, 3313 jobs, exit 0. Axiom audit via 'lake env lean' on skalse_theorem1, skalse_theorem3_only_if, skalse_corollary3, ng_vstar_shaped, ng_qstar_shaped and ng_shaping_preserves_optimal reports only [propext, Classical.choice, Quot.sound] for each; no sorryAx. RELATED rather than EXACT/EQUIVALENT: a reward-hacking characterization is not a formalization of the Goodhart/Strathern statement this row records. Not usable as atlas coverage without resolving version skew (atlas pins Lean v4.33.0 and Mathlib db584cd6d46c92f209a44c0f1c829460d327499d). Companion paper (Sheridan 2026) is self-published and not peer reviewed; maintenance is unproven. The MDP/Bellman/shaping layer is a candidate Lean-native substrate for BY-039." }, { "repository": "https://github.com/RichardZandi/Impossibility", "revision": "3e951593660655055953330226039b5c79c6b09a", "framework": "Lean", "license": "NONE", "declaration": "goodhart_impossibility (Impossibility/Goodhart/GoodhartUCI.lean)", "inspection_state": "SOURCE_INSPECTED", "relationship_review": "DISTINCT", "notes": "Rejected. The lemma states that a computable, extensional classifier separating two programs is contradictory, which is Rice's theorem under a Goodhart name; it encodes no proxy/target divergence. Rice is already covered at BY-012. No license file, so it could not be reused even if the statement matched." }, { "repository": "https://github.com/paulklemstine/Lean", "revision": "620407cde0610ab76ba3e055ef143b341a8e6cd9", "framework": "Lean", "license": "NONE", "declaration": "goodhart_divergence_exists (Catalog/Computation/GoodhartsRepulsor.lean)", "inspection_state": "SOURCE_INSPECTED", "relationship_review": "DISTINCT", "notes": "Rejected as vacuous. The hypothesis is 'exists s, proxy increases and true value decreases' and the conclusion is 'exists s, true value decreases', a weakening of the hypothesis; the theorem carries no Goodhart content. A neighbouring theorem in the same file proves set intersection is a subset of one operand. No license file; the repository duplicates generated copies of the same file across several directories." }, { "repository": "https://github.com/JonRademacher/standing-algebra", "revision": "df1fb08f32544de796f9a90a4a8d102f1a57635f", "framework": "Lean", "license": "Apache-2.0", "declaration": "Collapse_Demonstrations.Goodhart_Drift.GoodhartDrift (Collapse_Demonstrations/Goodhart_Drift/Goodhart.lean)", "inspection_state": "SOURCE_INSPECTED", "relationship_review": "DISTINCT", "notes": "Rejected. The file supplies only a predicate GoodhartDrift asserting that some pair of times exists at which a proxy rises while true value falls, and proves nothing about it; the file states its own purpose as showing the notion is satisfiable. A definition without a theorem is not a formalization of this row." } ], "tags": [ "agent-incentives" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "RETRIAGED 2026-09-13, from CANDIDATE_LEAD, and the verdict changed because both of this row's remaining reading tasks turn out to be done. FIRST, THE CANDIDATES ARE ALL ADJUDICATED. The previous note said 'The other three candidates remain unadjudicated, so the row stays a reading task'; that is false on this row's own records. Every one of the four candidate_formalizations entries carries an inspection_state and a relationship_review: RichardZandi/Impossibility is SOURCE_INSPECTED and DISTINCT (Rice's theorem under a Goodhart name, already covered at BY-012, and no licence), paulklemstine/Lean is SOURCE_INSPECTED and DISTINCT (the conclusion is a weakening of the hypothesis, so the theorem carries no Goodhart content, and no licence), JonRademacher/standing-algebra is SOURCE_INSPECTED and DISTINCT (a predicate with nothing proved about it), and audieleon/goodhart is REPRODUCED and RELATED but is a formalization of SKALSE, whose paper is graded in section 26 of docs/provenance/source-coverage-audit.md at 0 Yes for this atlas. SECOND, THE STATEMENT SOURCES ARE READ. Both were supplied by the maintainer and read on 2026-09-10 from rendered page images, and are manifested with sha256 in the private manifest of 2026-09-10: strathern-published-eur-rev-1997-...pdf, sha256 30ef68d3..., journal page 308; and goodhart-published-macmillan-1984-...pdf, sha256 86e69772..., book page 96 and Introduction pages 9 and 12. WHAT THE READING SETTLED. The famous sentence is Strathern's own coinage at journal page 308 - she does not quote Goodhart and attributes the NAME to Hoskin - so this row is right to name her as its source. Her only quantitative content is loss of resolution in ONE measure: 'The more a 2.1 examination performance becomes an expectation, the poorer it becomes as a discriminator of individual performances.' That is not proxy-target divergence and is not the shape of any candidate: Skalse unhackability is a disagreement between two orderings and Zhuang and Hadfield-Menell's Theorem 1 is about omitted attributes running to their floor. Goodhart's own sentence, at book page 96, is a third claim again - 'any observed statistical regularity will tend to collapse once pressure is placed upon it for control purposes' - and his Introduction at page 12 identifies that chapter as the law's origin while the name was applied afterwards. SO A FRESH FORMALIZATION IS OWED, AND ITS SHAPE IS NAMED: a measure whose image or fibres coarsen as it is adopted as a target, with an optimization dynamic giving 'becomes an expectation' a referent. AISafetyAtlas.Sovereignty.Arena.collapse and AISafetyAtlas.Oversight.VarietyBound.collapse are the nearest objects in the tree and are a CANDIDATE ARROW, NOT AN ESTABLISHED ONE - without the dynamic it is a resemblance, and per docs/provenance/resemblance-mapping-is-not-implication it is recorded as a lead rather than as coverage. NO COVERAGE-AUDIT SECTION, AND THAT IS A DECISION. Strathern's paper is a seventeen-page anthropological lecture and Goodhart's is a 288-page book; the atlas's interest is one sentence in each, both quoted above and in the manifest. Grading the remainder statement by statement would be grading a subject this ledger does not have, so the readings are recorded here and in the manifest and not as audit sections. WHAT THE PREVIOUS NOTE SAID AND STILL HOLDS: four candidate formalizations are catalogued on this row -- a Skalse Theorem 1 development with a potential-shaping layer, and three further Goodhart encodings. Corrected 2026-09-10: this said 'None has been adjudicated for statement match, licence or reproduction', which stopped being true when docs/provenance/by037-by038-goodhart-campbell-plan.md D4 adjudicated audieleon/goodhart at revision 29128f3f9bcafb30d019682b63c1b582bcadf7b9 on all three counts - statement by statement in docs/provenance/external-formalizations.md, licence as Apache-2.0 with no upstream NOTICE so sections 4(a) to 4(c) apply and 4(d) does not, and reproduced by cloning twice. Its verdict is cite, do not vendor. That adjudication is scoped to the tree's Skalse files; the paper itself is now graded, in section 26 of docs/provenance/source-coverage-audit.md, at 0 Yes, 1 Partial, 15 No -- no printed claim of it is covered here, because its object is the orderings two reward functions induce on a policy set rather than optimization of a proxy, and the substrate it needs is an occupancy embedding with value as a linear functional of the reward, which this tree does not have; its proofs/GoodhartProofs/MDP/ subtree has not been assessed and is the nearest external match to AISafetyAtlas.Decision.DiscountedValue. SUPERSEDED 2026-09-20 on the substrate point only: the occupancy embedding and value as a linear functional of the reward now exist, as AISafetyAtlas.Decision.Occupancy, and section 26 regrades to 5 Yes, 0 Partial, 13 No with Definitions 1 and 2 rendered in AISafetyAtlas.Goodhart.Hackability under LAND-GOODHART-HACKABILITY-001. That does NOT move this row: Skalse unhackability is still not Strathern's sentence, and what BY-037 owes is still a measure whose fibres coarsen under adoption as a target. That adjudication is the strongest of the four; the other three are rejected above. Updated 2026-09-11: Lean now exists on this row's subject, though NOT on this row's claim. AISafetyAtlas.Goodhart.Regressional and AISafetyAtlas.Goodhart.Extremal are built and hosted by LAND-GOODHART-SELECTION-001, and Manheim and Garrabrant is graded in section 17 of docs/provenance/source-coverage-audit.md. That does not cover BY-037: those theorems are atlas-original work on Manheim and Garrabrant's MODEL, and this row's claim is Strathern's sentence, and survey-ref-074 and survey-ref-075 are now both read. The verdict is no longer CANDIDATE_LEAD: as of 2026-09-13 Strathern's own text has been read, survey-ref-074 and survey-ref-075 are both held and quoted, and what is owed is a build and not a reading." } }, { "id": "BY-038", "name": "Campbell's law", "paper_reference": "Table 1", "survey_proof_assessment": "NOT_PROVEN_IN_SURVEY", "informal_claim": "Heavy decision use of a quantitative indicator creates pressure that corrupts the indicator and represented process.", "original_source_refs": [ "survey-ref-076" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "campbell's law", "campbell law", "regression to the mean" ] }, "tags": [ "agent-incentives" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "RETRIAGED 2026-09-13, from CANDIDATE_LEAD. The sole named candidate is adjudicated and the statement source is read, so what is owed is a build rather than a reading. THE CANDIDATE, READ 2026-09-10: Ashton, arXiv:2011.01010v2, sha256 7595f171..., 7 pp., manifested 2026-09-10. It carries NO NUMBERED STATEMENTS - its content is a dog-barometer toy MDP - and it classifies its own problem as Causal Goodhart in Manheim and Garrabrant's scheme, so it is not a formalization of this row's claim and is evidence for fusing the two rows rather than against. Manheim and Garrabrant themselves contain no theorem, proposition, lemma or corollary of any kind, and their section 4 row named Campbell's Law keeps only the indicator-corruption half. THE SOURCE, READ 2026-09-10 from a rendered page image: campbell-reprint-jmde-2011-assessing-the-impact-of-planned-social-change.pdf, sha256 8e5c0735..., at the reprint's page 34. TWO LAWS, NOT ONE: 'The more any quantitative social indicator is used for social decision-making, the more subject it will be to corruption pressures and the more apt it will be to distort and corrupt the social processes it is intended to monitor', and print calls them 'these two laws' in the next sentence, illustrating them with Skolnick's clearance-rate work which shows 'both corruption of the indicator itself and a corruption of the criminal justice administered'. The SECOND conjunct - the monitored process is deformed - is what keeps this row distinct from BY-037 and is what a fresh formalization is owed. A VERSION CAVEAT THAT MUST NOT BE LOST: the held file is the 1976 Occasional Paper as reprinted in the Journal of MultiDisciplinary Evaluation 7(15), February 2011, doi 10.56645/jmde.v7i15.297, CC BY-NC 4.0. survey-ref-076 cites the 1979 Evaluation and Program Planning article. Same text by report, different container, and the two have NOT been compared; the 1979 version is not held. PRINT'S OWN HEDGE, RECORDED: the laws are offered 'at least for the U.S. scene' and illustrated with evidence print calls 'predominantly anecdotal'. They are empirical generalizations, not theorems, so no formalization can match them by statement; what a candidate can match is a MODEL of them, which is why the requirement stated below is a requirement on the model. NO COVERAGE-AUDIT SECTION, AND THAT IS A DECISION: the paper is forty-one pages on quasi-experimental evaluation of social programmes and the atlas's interest is one section of it, quoted here and in the manifest; grading the rest would be grading a subject this ledger does not have. WHAT THE 2026-09-11 TRIAGE ESTABLISHED, UNCHANGED: the previous note said only that nobody had read the source against the tree; that is no longer where this row stands. THE REQUIREMENT IS STATED: docs/provenance/by037-by038-goodhart-campbell-plan.md D2 records that any candidate for Campbell's law must carry an INTERVENTIONAL layer distinguishing the process before and after the indicator enters the decision rule, since the claim is that measuring corrupts, not merely that a proxy and a goal come apart. THE SUBSTRATE NOW EXISTS: that plan said to propose the row after the causal layer merged, not before, naming AISafetyAtlas.Causal.StructuralModel and AISafetyAtlas.Causal.Incentive; both merged on 2026-09-10, so the precondition is met and the row is proposable. THE CANDIDATE IS NAMED AND PINNED: Ashton, arXiv:2011.01010v2, sha256 7595f171e54f9a12be1c1803350b66eb1d64a88adf613021f50d596ba91086ab, the only paper named for this row. It carries no numbered statements - its content is a dog-barometer toy MDP in which an off-policy learner is fooled by an intervention on a predictive variable and an on-policy learner is not - and it classifies its own problem as Causal Goodhart, metric manipulation, in Manheim and Garrabrant's scheme. Per D1 the two rows nonetheless stay distinct. PRINT EVIDENCE ADDED 2026-09-11: Manheim and Garrabrant's own section 4 carries a row named Campbell's Law with equation (7), M_R = G_R + X and M_A = G_A . X, asserting that the correlation between the agent's goal and the agent's metric is zero over the full state set but positive on the subspace the regulator selects. Section 17 of docs/provenance/source-coverage-audit.md grades that No and names it the sharpest uncovered claim in that paper; it is quantitative and provable and needs a second actor and a product-of-two-metrics carrier the Goodhart cluster does not have. That is the concrete shape a candidate would take. Still uncovered. survey-ref-076 is the statement source and is now read, in the container named above." } }, { "id": "BY-039", "name": "Reward corruption unsolvability", "paper_reference": "Table 1", "survey_proof_assessment": "PROVEN", "informal_claim": "No agent can in general distinguish all reward-channel corruption from genuine reward without additional assumptions.", "original_source_refs": [ "survey-ref-077" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Wireheading.CRMDP", "declaration": "AISafetyAtlas.Wireheading.CRMDP.Model.everitt_theorem_eleven", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Wireheading.CRMDP AISafetyAtlas.Examples.SixTargets", "scope_delta": { "summary": "One fixed deterministic transition rather than a class ranging over stochastic kernels; extrema supplied as structure fields rather than derived; a continuous reward interval rather than the source's finite uniform reward grid.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Wireheading.CRMDP.Model.everitt_theorem_eleven", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Wireheading.CRMDP.Env.observed_complement", "AISafetyAtlas.Wireheading.CRMDP.return_add_complement", "AISafetyAtlas.Wireheading.Corruption.ComplementedClass.everitt_theorem_eleven" ], "application": "Reward corruption: an agent optimizing an observed reward channel cannot be guaranteed to avoid half-maximal regret against the true reward. Theorem 11 of Everitt, Krakovna, Orseau, Hutter and Legg (2017), over a corrupted-reward MDP." } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Two levels. CRMDP.Model.everitt_theorem_eleven is the canonical declaration and is stated for a model that carries states, actions, one fixed deterministic transition, unit-interval true and observed rewards, a corruption channel, source-shaped action/observation histories, policies that read them, and finite-horizon return. Inside it the source construction is verified rather than assumed: Env.observed_complement proves that an environment and its complement induce the same observed reward, run_complement proves that no policy can therefore separate them, and return_add_complement proves the source equation (3) that true returns sum to the horizon. Completeness with respect to true rewards and corruption channels is structural, since Env is their full bounded function-space product and is closed under complement with no side condition; unlike the source's complete CRMDP class, one Lean Model does not range over transition kernels. The reward bound is load-bearing: the superseded unrestricted-real interface combined with an attained worst-environment field forced every inhabitant to have zero regret by a scaling argument. Examples.SixTargets.nonzeroCRMDPModel_worstCaseRegret now proves worst-case regret exactly one in a concrete two-state, two-action model. Corruption.ComplementedClass.everitt_theorem_eleven remains as the algebraic core lemma that the model instantiates, and it supplies the certificate consumed by AISafetyAtlas.Preference.Regret. RELATED, not EXACT: one fixed deterministic transition replaces a class ranging over stochastic kernels and induced measures on histories; the attained best policy, worst environment and worst policy are structure fields rather than derived from finiteness; and rewards range over the continuous interval [0,1] rather than a finite uniform grid. WHERE EACH OF THOSE IS CLOSED, as of 2026-09-10, though not jointly and not in this declaration: AISafetyAtlas.Wireheading.RewardGrid.everitt_theorem_eleven_gridClass carries print's uniform grid and derives all three extrema; CRMDP.StochModel.everitt_theorem_eleven replaces the transition with a PMF and the return with an expectation, which at print's finite state set is print's kernel exactly; and CRMDP.MixedModel.everitt_theorem_eleven adds print's own quantifier over a possibly stochastic policy, so the return averages over the agent too. CRMDP.Model.toStoch_returnValue and CRMDP.StochModel.toMixed_returnValue chain the three to the same numbers. The grade stays RELATED because no single rendering has all four printed features at once: the grid-class one is deterministic in dynamics and policy both, and the two stochastic ones take their extrema as fields over the interval carrier. Deriving the extrema over a policy class of distributions is the affine-functional argument named in RewardGrid's header and is still not carried out.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "isabelle-afp" ], "query_terms": [ "reward corruption", "corrupted reward channel", "reward tampering", "corrupt reward mdp", "reward misspecification", "markov decision process" ] }, "tags": [ "agent-incentives" ], "public": { "group": "Preferences, rewards and incentives", "summary": "If the reward can be faked, the agent cannot tell it is being fooled.", "use": "Reward tampering.", "title": "Reward corruption", "attribution": "Everitt, Krakovna, Orseau, Hutter & Legg" } }, { "id": "BY-040", "name": "Uncontrollability of AI", "paper_reference": "Table 1", "survey_proof_assessment": "PROOF_SKETCH", "informal_claim": "Perfect explicit control of advanced AI fails in the source's degenerate self-referential conditions.", "original_source_refs": [ "survey-ref-008" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "uncontrollability of ai", "ai control" ] }, "tags": [ "control-theory", "verification" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "RETRIAGED 2026-09-13, from UNTRIAGED. THE SOURCE IS NOW HELD AND THE DECIDING PAGE IS READ. The file is yampolskiy-arxiv-v1-2020-on-controllability-of-ai.pdf, sha256 57abe0f8bc5d5eb30d4bad0da76cf48eb3b2c399474f50c0d54b4ed3c4233eaf, 59 pp., fetched 2026-09-13. ITS OWN TITLE IS 'On Controllability of AI', not the title survey-ref-008 cites; the filename follows the document. THE DECIDING FACT, read from a rendered image of page 32: the ONLY thing in fifty-nine pages that is labelled a theorem is a BLOCK QUOTATION of somebody else's - 'Theorem 1. The harming problem is undecidable', quoted verbatim from reference [126], which is Alfonseca et al., together with their proof and their corollary. Print states no result of its own anywhere; it is a survey of arguments and quotations for the claim that advanced AI cannot be fully controlled. SO THIS ROW AND BY-025 SHARE THEIR ONLY THEOREM, AND IT BELONGS TO BY-025. That is now checked rather than suspected, and it is the reason this row is graded distinct: a formalization of the quoted theorem would be coverage of BY-025, not of this row, and this row's own claim - perfect explicit control fails in the source's degenerate self-referential conditions - has no printed statement behind it here. A FRESH FORMALIZATION IS OWED AND IT WOULD BE ATLAS-ORIGINAL. Anyone building it must not cite this paper for a theorem. The tree's nearest objects are AISafetyAtlas.Sovereignty's effectivity layer and AISafetyAtlas.Control, and neither is indexed by self-reference. VOCABULARY NOTE, so the verdict is not read as more than it is: the statability vocabulary has no value for a source that states no theorem of its own. TRIAGED_DISTINCT is the nearest, and its gloss's first clause - candidates were examined and found distinct - is VACUOUS here rather than satisfied, because the sweep examined nothing: there is no printed statement for a candidate to be distinct from. Its second clause, that a fresh formalization is owed, does hold and is what this note above says. UNTRIAGED is wrong, not weaker: its gloss is that nobody has compared the source to the tree, and that comparison is now done. Whether the vocabulary gains a value for this case is a maintainer decision and is not taken here." } }, { "id": "BY-041", "name": "Impossibility of unambiguous communication", "paper_reference": "Table 1", "survey_proof_assessment": "CONDITIONAL", "informal_claim": "Unambiguous communication is impossible under the source's strict assumptions about language and interpretation.", "original_source_refs": [ "survey-ref-078" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Statement-level equivalence to each cited source remains subject to source comparison.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "unambiguous communication", "ambiguity communication" ] }, "tags": [ "information-theory", "multi-agent" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged. THE SOURCE IS NOT HELD, and the ledger's locator is a ResearchGate DOI rather than a publisher, a repository or an author page - which is why no lawful fetch was attempted on 2026-09-13 beyond recording the route. Howe and Yampolskiy, Impossibility of Unambiguous Communication as a Source of Failure in AI Systems. ONE THING ALREADY CHECKABLE: the tree has ambiguity-shaped objects - AISafetyAtlas.Control and the joint-observation residual modules under Oversight use the word in the sense of a map failing to determine its argument, which is the AISafetyAtlas.Knowledge.Knowable shape. Whether print's ambiguity is that one, or a linguistic notion with no single-valuedness reading, is exactly what the paper would settle and is not guessed at here." } }, { "id": "BY-042", "name": "Unfairness of explainability", "paper_reference": "Table 1", "survey_proof_assessment": "PROOF_SKETCH", "informal_claim": "A verifier and a decision-maker have structurally unequal strategic positions when explanations omit the full execution trace.", "original_source_refs": [], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Introduced in the survey; statement-level semantic review is required.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "unfairness of explainability", "strategic inequality" ] }, "tags": [ "interpretability", "multi-agent" ], "statability": { "verdict": "UNTRIAGED", "note": "Note rewritten 2026-09-13; verdict unchanged, and the reason is a LEDGER DEFECT rather than a fetch failure. THIS ROW HAS NO original_source_refs AT ALL - checked 2026-09-13 - so there is no source to read and no statement to compare with the tree. Every other UNTRIAGED row names at least one work. Until a source is attached, the row asserts an informal claim - that a verifier and a decision-maker have structurally unequal strategic positions when explanations omit information - with nothing behind it. The nearest tree objects are the Debate protocol under AISafetyAtlas/Upstream and the Oversight cluster, both of which are about asymmetric information between a checker and a proposer; those are candidates for a future statement source and are NOT evidence that this row's claim is covered. THE ACTION THIS ROW NEEDS IS NOT A READING: it is a decision by the maintainer about which source the survey meant, or a decision that the row carries no source and should say so." } }, { "id": "BY-043", "name": "Misaligned embodiment", "paper_reference": "Table 1", "survey_proof_assessment": "PROOF_SKETCH", "informal_claim": "Mistakenly cloned self-interested agents cannot perfectly control one another in the survey's model.", "original_source_refs": [ "brcic-yampolskiy-2023-own-results" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Compositional.Symmetry", "declaration": "AISafetyAtlas.Compositional.Symmetry.Protocol.no_unique_leader_from_symmetric_start", "relationship": "RELATED", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Compositional.Symmetry", "scope_delta": { "summary": "The Angluin-style inductive core only, over an assumed symmetric configuration: simplified port routing, no full covering theory, no randomized symmetry breaking. Do not infer BY-043 from the A3 dependency alone.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "Fleet", "evaluating_one_covers_its_peers", "no_designated_agent_emerges", "symmetry_is_the_shared_cause" ], "module": "AISafetyAtlas.Compositional.AgentNetwork", "build_command": "lake build AISafetyAtlas.Compositional.AgentNetwork", "scope_delta": { "summary": "BRIDGE module over Angluin's anonymous-network results, read as a deployed fleet of identical agents. Both halves: the symmetry that lets one evaluation cover every agent with the same view is the symmetry that stops any agent becoming the designated coordinator, so an architecture gets the cheap evaluation and the elected monitor together or not at all. Witnessed at a two-agent fleet in AISafetyAtlas.Examples.Scenario.Practitioner.", "evidence": "docs/provenance/practitioner-checkers.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Compositional.Symmetry.Protocol.no_unique_leader_from_symmetric_start", "type": "NEW_PROOF", "source_declarations": [], "application": "Multi-agent symmetry: no anonymous deterministic protocol elects a unique leader from a symmetric configuration. Angluin's 1980 symmetry obstruction, proved in-tree over an explicit port-labelled network model." }, { "atlas_declaration": "AISafetyAtlas.Compositional.AgentNetwork.symmetry_is_the_shared_cause", "type": "BRIDGE", "source_declarations": [ "Fleet", "evaluating_one_covers_its_peers", "no_designated_agent_emerges" ], "application": "Evaluating and architecting a fleet of identical agents. Agents whose local view of the deployment agrees to depth n hold the same state after n rounds, so an evaluation transfers along the symmetry and a large fleet need not be tested instance by instance. The same fact means no run of any deterministic anonymous protocol ends with exactly one agent in a distinguished role, so a safety design that names one instance the monitor or the kill-switch holder is relying on an identifier, a topology asymmetry or an external assignment -- and naming which is the obligation. A fleet whose instances carry distinct keys is not an instance of this, which is the useful reading.", "review_status": "STATEMENT_REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": false, "evidence": "docs/interpretation-reviews/review-symmetry-is-the-shared-cause.md" } } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Lean formalizes the Angluin-style inductive core (Angluin, Local and Global Properties in Networks of Processors, STOC 1980): identical deterministic code plus indistinguishable observations preserves a symmetric configuration, so at least two nodes cannot reach exactly one leader in finitely many synchronous rounds. The network beneath it is now formalized separately in AISafetyAtlas.Compositional.Networks, catalogued as LAND-ANGLUIN-001, where observational indistinguishability is derived from port structure rather than assumed: runFor_eq_of_view_eq is the depth-n view lemma and no_unique_leader_of_fixedPointFree gives the automorphism route. That work deliberately does not change this row. RELATED only: this is a reusable dependency and source-backed repair direction, not a formalization of the survey's self-interest, mistaken-belief, embodiment or mutual-control semantics. Randomness, identifiers and asymmetric observations are explicit exits.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [ "isabelle-afp" ], "query_terms": [ "misaligned embodiment", "operationally cloned", "clone misalignment", "symmetry breaking", "anonymous network", "leader election" ] }, "tags": [ "multi-agent", "compositionality" ], "public": { "group": "Aggregation and multi-agent structure", "summary": "The survey claims cloned agents cannot control each other. Only the symmetry result is proved.", "use": "The difference between a claim and a proof of it.", "title": "Misaligned embodiment", "attribution": "Brcic & Yampolskiy; Angluin for the formalized core" } }, { "id": "BY-044", "name": "Limited self-awareness", "paper_reference": "Table 1", "survey_proof_assessment": "PROOF_SKETCH", "informal_claim": "An agent cannot be perfectly self-aware across the survey's defined operational boundaries.", "original_source_refs": [ "brcic-yampolskiy-2023-own-results" ], "result_shape": "POINT_IMPOSSIBILITY", "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.SelfAwareness", "atlas_module": "AISafetyAtlas.SelfAwareness", "relationship": "EQUIVALENT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.SelfAwareness.Model", "AISafetyAtlas.SelfAwareness.AgentAware", "AISafetyAtlas.SelfAwareness.PerfectlySelfAware", "AISafetyAtlas.SelfAwareness.Model.not_aware_of_le", "AISafetyAtlas.SelfAwareness.Model.process_not_self_aware", "AISafetyAtlas.SelfAwareness.Model.not_agentAware_of_maximal", "AISafetyAtlas.SelfAwareness.Model.limited_self_awareness", "AISafetyAtlas.SelfAwareness.Model.not_perfectlySelfAware" ], "build_command": "lake build AISafetyAtlas.SelfAwareness AISafetyAtlas.Examples.SelfAwareness", "scope_delta": { "summary": "The process-compositional proof is mechanized over a finite fixed-horizon semilattice of available composites. The source's implicit non-cancellation semantics is unpacked as a strict additive awareness-cost law; propagation time and bounded lag are absorbed into the horizon-relative awareness relation rather than represented by a separate clocked model.", "evidence": "docs/provenance/limited-self-awareness.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.SelfAwareness.Model.not_aware_of_le", "type": "NEW_PROOF", "source_declarations": [], "application": "Complete process awareness: a constituent cannot actively observe and predictively model the composite containing that same awareness activity under strict positive awareness cost. Brcic and Yampolskiy (2023), Proposition 4.7 formal core." }, { "atlas_declaration": "AISafetyAtlas.SelfAwareness.Model.process_not_self_aware", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.SelfAwareness.Model.not_agentAware_of_maximal", "type": "NEW_PROOF", "source_declarations": [], "application": "Complete process awareness: every maximal available composite lacks an available internal observer; the source-facing existential theorem follows by finite maximality." }, { "atlas_declaration": "AISafetyAtlas.SelfAwareness.Model.limited_self_awareness", "type": "NEW_PROOF", "source_declarations": [], "application": "Bounded-horizon process self-awareness: some available atomic or composite internal process has no available complete observer. Brcic and Yampolskiy (2023), Theorem 4.8 process-compositional interpretation." }, { "atlas_declaration": "AISafetyAtlas.SelfAwareness.Model.not_perfectlySelfAware", "type": "NEW_PROOF", "source_declarations": [] } ] }, "ai_safety_relevance": "The survey identifies this result as relevant to limits on AI; no direct real-system implication is asserted here.", "ai_interpretation_status": "STATEMENT_REVIEWED", "interpretation_review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-08-12", "statement_reviewed": true, "interpretation_reviewed": false, "evidence": "docs/interpretation-reviews/review-by-044-selfawareness.md" }, "notes": "EQUIVALENT formalization of Definition 4.6, Proposition 4.7 and Theorem 4.8 in section 4.3, after first-author statement-level review on 2026-08-12. The Lean model separates the vertical strict-awareness-extension obstruction from the horizontal maximal-composite proof. Ordinary cycles are permitted and exhibited: two singleton processes can observe one another while their joint composite remains unobserved. The source's crucial non-cancellation step is exposed as awareness_cost, requiring the observer-target composite to cost at least one positive awareness increment more than the target. This is an explicit formal unpacking of awareness as distinct active observation and predictive modelling that is itself positive-cost computation within the bounded lag, not an additional physical premise. available_finite records the bounded-horizon resource consequence directly, and timing assumptions 4-5 are semantic in aware rather than represented by a separate clock. The executable example also shows that irreflexivity alone does not imply the theorem without composite closure. EQUIVALENT rather than EXACT records the semilattice and fixed-horizon representational choices. No consciousness, Lawvere, Wolpert, Breuer, selected-property, approximate, delayed, or thermodynamic claim is made. Source map and fidelity residual: docs/provenance/limited-self-awareness.md.", "formal_library_search": { "searched_on": "2026-07-28", "evidence_file": "docs/provenance/formalization-search.json", "searched_corpora": [ "mathlib", "isabelle-afp", "rocq-undecidability", "hol4", "hol-light", "agda-stdlib" ], "candidate_corpora": [], "query_terms": [ "limited self-awareness", "locus of self", "self-awareness limit" ] }, "tags": [ "interpretability" ], "public": { "group": "What an observer can recover", "summary": "A bounded agent cannot completely monitor every recursively composable internal process under strict positive awareness cost.", "use": "The process-compositional version of a self-awareness limit; cycles are allowed.", "title": "Limited self-awareness", "attribution": "Brcic & Yampolskiy" }, "escape_routes": [ { "axis": "ADD_RESOURCE", "status": "NAMED_ONLY", "note": "The bounded resource the statement holds fixed is the awareness horizon, recorded directly as Model.available_finite - the module's non-claims say it is the fixed-horizon consequence of the source's bounded budget and positive minimum process cost, not derived here from a scalar budget. Dropping it is a real escape from Theorem 4.8 and therefore from the corollary: limited_self_awareness obtains its target from available_finite.exists_maximal, and with no maximal available composite that step has nothing to choose. It is graded NAMED_ONLY because no declaration in this tree states what happens in an unbounded horizon. That grade is about the tree and not about what is known: an adversarial review on 2026-09-07 exhibited the escape outside the tree, keeping every Model field except available_finite - the naturals under max, available = univ, aware o t := t < o, cost = id, minAwarenessCost = 1 - where every remaining field holds and every target has the observer t+1. So dropping the horizon does not merely stop this argument, it makes the conclusion false. Upgrading the grade needs a Model-minus-finiteness structure in the tree, which is a design decision about what the module's model class is and not a formality, so the route stays NAMED_ONLY until someone rules on it. Two things this route does not reach. The vertical obstruction survives it: not_aware_of_le and process_not_self_aware use only awareness_cost and minAwarenessCost_pos, so a constituent still cannot completely model its containing composite however large the horizon. And the strict positive awareness cost itself is the model's defining premise rather than a resource it holds fixed - weakening it abandons the source's reading of awareness as distinct active observation instead of escaping the theorem - so it is not recorded as a route on any axis in this vocabulary." } ] }, { "id": "CLM-WOLPERT-KNOW-001", "name": "Physical knowledge by inference devices", "informal_claim": "A device physically knows a realized target value over a set W when one selector of realized setup blocks answers every target probe correctly and has the required nonempty true/false intersections with W. This entails weak inference; on a W that refines a Boolean target it is truthful and cannot affirm both values; and every device has a Boolean target it physically knows at no value over any W.", "original_source_refs": [ "survey-ref-054" ], "related_result_ids": [ "BY-024", "LAND-KNOW-001" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Inference.PhysicalKnowledge", "atlas_module": "AISafetyAtlas.Inference", "relationship": "EQUIVALENT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Inference.ImageValue", "AISafetyAtlas.Inference.SetupBlock", "AISafetyAtlas.Inference.PhysicalKnowledgeWitness", "AISafetyAtlas.Inference.PhysicallyKnows", "AISafetyAtlas.Inference.PhysicalKnowledgeWitness.weaklyInfers", "AISafetyAtlas.Inference.PhysicallyKnows.weaklyInfers", "AISafetyAtlas.Inference.RefinesOn", "AISafetyAtlas.Inference.PhysicalKnowledgeWitness.eq_target_of_mem_knownBlock", "AISafetyAtlas.Inference.PhysicalKnowledgeWitness.eq_target_on_of_refinesOn", "AISafetyAtlas.Inference.PhysicalKnowledgeWitness.negate", "AISafetyAtlas.Inference.PhysicallyKnows.negate", "AISafetyAtlas.Inference.physicallyKnows_false_iff_not_true", "AISafetyAtlas.Inference.true_on_of_physicallyKnows_true", "AISafetyAtlas.Inference.false_on_of_physicallyKnows_false", "AISafetyAtlas.Inference.not_physicallyKnows_true_and_false", "AISafetyAtlas.Inference.exists_never_physicallyKnown" ], "build_command": "lake build AISafetyAtlas.Inference.PhysicalKnowledge AISafetyAtlas.Examples.Inference.PhysicalKnowledge; python3 scripts/check_print_axioms.py", "scope_delta": { "summary": "Source-facing transcription of Wolpert 2018 Definition 11, the immediate weak-inference consequence, Lemma 17, Proposition 18, Corollary 19 and Corollary 22. The source's image Gamma(U) is a subtype of realized target values; its setup partition is represented by realized setup labels, which label exactly the same fibres. Bool renames the source's two-valued conclusion set, and clause (i) is expressed as Y(u)=true iff Gamma(u)=gamma rather than by a Kronecker probe, avoiding a decidable-equality restriction. No finiteness, probability, topology or physical-worldline structure is added. This row covers only the named first cluster, not the remainder of the 2018 paper.", "evidence": "docs/provenance/wolpert-2018-knowledge.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Inference.PhysicallyKnows.weaklyInfers", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.PhysicalKnowledgeWitness.eq_target_on_of_refinesOn", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.physicallyKnows_false_iff_not_true", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.not_physicallyKnows_true_and_false", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.exists_never_physicallyKnown", "type": "NEW_PROOF", "source_declarations": [] } ] }, "ai_safety_relevance": "A formal vocabulary for device-relative knowledge and its limits. Applying it to an AI system requires an explicit model of worlds, configuration, conclusion, and the context set W; the Lean definition alone supplies none of those interpretations.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "EQUIVALENT transcription of the first physical-knowledge cluster in Wolpert 2018, read from arXiv:1711.03499v3. The encoding preserves the selector over realized target values, the setup-partition blocks, the global correctness clause and both nonempty W-intersection clauses. The executable Bool-pair example witnesses all Definition 11 obligations. This predicate is not Knowledge.Knowable: physical knowledge chooses a setup block per target probe, whereas Knowable requires one decoder uniformly over all states. No bridge between those notions, no S4/S5 claim, and no actual-device interpretation is asserted. Corollaries 20-24 are owned by CLM-WOLPERT-EPISTEMIC-001; the defects in Corollaries 21(ii) and 25 are owned by LAND-WOLPERT-KNOW-DEFECTS-001. Inference complexity, Kraft and entropy remain outside this row.", "tags": [ "interpretability" ], "public": { "group": "What an observer can recover", "summary": "Knowing a fact requires more than producing the right bit: one selected configuration per alternative must answer correctly, and the claimed case must actually occur in context.", "use": "A precise device-level account of knowledge, truthfulness and an unavoidable unknown target.", "title": "Physical knowledge", "attribution": "Wolpert" } }, { "id": "CLM-WOLPERT-EPISTEMIC-001", "name": "Epistemic consequences of physical knowledge", "informal_claim": "For Wolpert's physical-knowledge predicate, knowledge is truth-preserving under the stated context-refinement hypotheses but is not logically omniscient: known premises can force a consequence to be true without making it known. Distinguishability also prevents mutual knowledge of conclusions and, under the Corollary 24 hypotheses, forces at least three inequivalent binary targets to remain unknown.", "original_source_refs": [ "survey-ref-054" ], "related_result_ids": [ "CLM-WOLPERT-KNOW-001", "BY-024", "LAND-WOLPERT-KNOW-DEFECTS-001" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Inference.PhysicalKnowledge.Epistemic", "atlas_module": "AISafetyAtlas.Inference", "relationship": "EQUIVALENT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Inference.boolImplies", "AISafetyAtlas.Inference.TrueOn", "AISafetyAtlas.Inference.corollary20_i", "AISafetyAtlas.Inference.corollary20_ii", "AISafetyAtlas.Inference.corollary20_iii", "AISafetyAtlas.Inference.corollary20_iv", "AISafetyAtlas.Inference.corollary21_i", "AISafetyAtlas.Inference.FunctionallyEquivalent", "AISafetyAtlas.Inference.exists_three_inequivalent_not_weaklyInfers", "AISafetyAtlas.Inference.corollary23", "AISafetyAtlas.Inference.corollary24" ], "build_command": "lake build AISafetyAtlas.Inference.PhysicalKnowledge.Epistemic AISafetyAtlas.Examples.Inference.PhysicalKnowledge.Epistemic; python3 scripts/check_print_axioms.py", "scope_delta": { "summary": "Source-facing transcription of Wolpert 2018 Corollaries 20(i-iv), 21(i), 23 and 24, together with the earlier Corollary 3 engine used by Corollary 24 and an executable version of Example 9. Bool replaces the source's two truth values; Corollary 20(iv) is zero-indexed. The source assumes countable U for this epistemic subsection, but these proofs do not use countability, so Lean is more general. Every W-refinement and distinguishability premise is explicit. This row deliberately excludes the false printed Corollary 21(ii) and Corollary 25, which are separately machine-audited, and makes no blanket S4 or S5 claim.", "evidence": "docs/provenance/wolpert-2018-knowledge.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Inference.corollary20_ii", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.exists_three_inequivalent_not_weaklyInfers", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.corollary23", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.corollary24", "type": "NEW_PROOF", "source_declarations": [] } ] }, "ai_safety_relevance": "Separates truth from knowledge in a device model: deductive consequences may hold in the relevant context without becoming physically known. Applying this to an AI requires a separate model of worlds, setups, conclusions and the context W.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "EQUIVALENT only for the named sound cluster, not for Wolpert 2018 as a whole. The executable Example 9 model proves that distribution fails even while both premises are physically known. The existing row/column device model inhabits Corollary 3, and a source-faithful physical-knowledge certificate inhabits the full Corollary 24 premise. Corollary 21(ii) and Corollary 25 are not hidden inside this grade: the first has a valid repaired theorem plus countermodel, and the second is audited under the paper's own equation (11).", "tags": [ "interpretability", "provability-logic" ], "public": { "group": "Limits of self-knowledge and reflection", "summary": "Knowing premises can make a conclusion true without making it known.", "use": "Separating truthful reasoning from logical omniscience in embedded devices.", "title": "Physical knowledge is not omniscience", "attribution": "Wolpert" } }, { "id": "CLM-WOLPERT-APPROX-001", "name": "Bounds on inference accuracy, and the collapse of the exact limits", "informal_claim": "A target attaining at least three values is weakly inferred by some device, which is the hypothesis the 2008 corollary omits and the 2018 restatement supplies. Covariance accuracy is bounded below by a term depending only on the number of target values times the device's inference power, so for a binary target it is never negative. Under that accuracy both the mutual-inference theorem and the strong-inference inheritance theorem hold only exactly: two setup-distinguishable devices can each infer the other to within any positive epsilon, and a device can strongly infer another that infers a target almost perfectly while itself inferring that target with accuracy exactly zero.", "original_source_refs": [ "survey-ref-054" ], "related_result_ids": [ "BY-024", "CLM-WOLPERT-KNOW-001" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "module": "AISafetyAtlas.Inference.Existence", "atlas_module": "AISafetyAtlas.Inference", "relationship": "EQUIVALENT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Inference.identityDevice", "AISafetyAtlas.Inference.identityDevice_weaklyInfers", "AISafetyAtlas.Inference.exists_weaklyInfers_of_three_values", "AISafetyAtlas.Inference.inferenceAccuracy_ge", "AISafetyAtlas.Inference.inferenceAccuracy_nonneg_of_card_eq_two", "AISafetyAtlas.Inference.fig5_stronglyInfers", "AISafetyAtlas.Inference.fig5_accuracy_dev1", "AISafetyAtlas.Inference.fig5_accuracy_dev2", "AISafetyAtlas.Inference.fig5_accuracy_gap", "AISafetyAtlas.Inference.fig6_distinguishable", "AISafetyAtlas.Inference.fig6_accuracy_dev1", "AISafetyAtlas.Inference.fig6_accuracy_dev2", "AISafetyAtlas.Inference.fig6_accuracy_near_one", "AISafetyAtlas.Inference.exists_distinguishable_accuracy_near_one", "AISafetyAtlas.Inference.StronglyInfersPair", "AISafetyAtlas.Inference.not_stronglyInfersPair_id", "AISafetyAtlas.Inference.exists_pair_not_stronglyInfers", "AISafetyAtlas.Inference.recursive_iff_refines", "AISafetyAtlas.Inference.PrefixFree", "AISafetyAtlas.Inference.prop11_of_independent", "AISafetyAtlas.Inference.exists_complexity_inversion", "AISafetyAtlas.Inference.inv_complexity", "AISafetyAtlas.Inference.inv_complexity'" ], "build_command": "lake build AISafetyAtlas.Inference.Existence AISafetyAtlas.Inference.Stochastic.Bounds AISafetyAtlas.Inference.Stochastic.Approximation AISafetyAtlas.Examples.Inference.Existence AISafetyAtlas.Examples.Inference.StochasticBounds AISafetyAtlas.Examples.Inference.StochasticApproximation; python3 scripts/check_print_axioms.py", "scope_delta": { "summary": "Wolpert 2018 Propositions 7(1), 8, 9 and 10 - the section III results that are not restatements of the 2008 paper. Proposition 7(1) is proved without either standing hypothesis it is printed under: countability of U is used nowhere, and |U| >= 2 follows from the target attaining three values. Its hypothesis is shown to be sharp rather than asserted to be: two values is exactly the 2008 Corollary 1(ii) countermodel. Proposition 8 is proved without assuming the realized image is nonempty, which follows from concl_surjective. Propositions 9 and 10 are proved sharper than printed - Figures 5 and 6 are transcribed state by state and the covariance accuracies are computed as exact identities in the figure's parameter, p and (1-6b)/(1-4b) respectively, so the printed 'arbitrarily close to' claims follow from algebra rather than from estimates, and the parameter achieving a given epsilon is named outright in each case. The Definition 9 accuracy carries the finite FinPMF and positive-fibre-mass conditions of the 2008 stochastic layer.", "excluded": "This row does not cover Wolpert 2018 Example 6, the sharpness construction for Proposition 8, or the weaker knowledge operator raised in prose after Example 9. Definition 7 is SPECIALIZED for finite rather than countable ranges and for the absence of a semi-measure type. Proposition 12 is REFUTED and is owned by LAND-WOLPERT-KNOW-DEFECTS-001. The complete accounting is the closed inventory in docs/provenance/wolpert-2018-knowledge.md, enforced over every source item by scripts/check_wolpert_2018_status_table.py." } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Inference.identityDevice_weaklyInfers", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.exists_weaklyInfers_of_three_values", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.inferenceAccuracy_ge", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.fig5_accuracy_gap", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Inference.exists_distinguishable_accuracy_near_one", "type": "NEW_PROOF", "source_declarations": [] } ] }, "ai_safety_relevance": "The impossibility result the survey cites for this cluster is a statement about exact inference. Proposition 10 measures what it costs to relax that: nothing about it survives approximation, so no claim about approximate or noisy observers may be inherited from it. Applying either result to an AI system still requires a separate model of universes, setups and conclusions.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "EQUIVALENT for the four named propositions only, not for Wolpert 2018 section III as a whole; the excluded items are listed in scope_delta and tracked row by row in the provenance inventory. Proposition 7(1) supplies the hypothesis missing from 2008 Corollary 1(ii), which BY-024 records as REFUTED: the author repaired his own corollary ten years later, and the atlas countermodel sits at exactly the excluded cardinality. That agreement is clash 20 and does not change the REFUTED grade, because the 2008 statement is still false as printed. Propositions 9 and 10 are stated beside the deterministic theorems they qualify rather than instead of them: weaklyInfers_of_stronglyInfers and not_infersDevice_both_of_distinguishable both still hold for the very devices of Figures 5 and 6.", "tags": [ "information-theory" ], "public": { "group": "What an observer can recover", "title": "Approximate inference escapes the exact limit", "summary": "Two observers that cannot each infer the other exactly can each infer the other to any accuracy short of certainty; an observer that can reproduce another observer inherits none of its accuracy; and a question with at least three answers always has some observer that answers it.", "use": "Fixes what the observation limits do and do not rule out once observation is noisy, gives the cardinality condition under which an inferring device is guaranteed to exist, and bounds how low inference accuracy can go.", "attribution": "Wolpert" } }, { "id": "LAND-WOLPERT-KNOW-DEFECTS-001", "name": "Countermodels to Wolpert 2018 epistemic claims", "tags": [ "interpretability", "provability-logic" ], "notes": "Machine-checked audit artifacts for two defective printed claims in arXiv:1711.03499v3. Corollary 21(ii)'s second disjunct is refuted by an inhabited Boolean model; corollary21_ii_repaired proves the valid successive-implication weakening. Equation (11) is transcribed literally as KnowledgeEvent. For the four-state Definition 11 observer it yields the universal event. Under the paper's standing two-valued-target convention that characteristic is inadmissibly constant, so Corollary 25 is not closed under its own construction; under the natural singleton-image extension implemented by Definition 11, the observer cannot know the universal event and positive introspection is false. These findings are not an EQUIVALENT claim for the defective source statements. Wolpert 2018 Proposition 12 is a third machine-refuted printed claim, added here: C_mu(Gamma; D) <= |Gamma| x H_mu(X) fails on a four-state model meeting every printed hypothesis, by exactly (ln 3)/2. prop12_gap is an identity, so the refutation does not rest on a numeric estimate. The printed proof's final step compares two sums term by term and needs mu(x) >= 1/|Gamma| at every setup value, which the statement does not assume. Recorded as clash 23. Wolpert 2018 Example 6 is a fourth machine-refuted printed claim. Its sharpness claim for Proposition 8's bound is false: the construction's central identity is correct and verified, but the step after it pulls the factor (2 - |Gamma(U)|)/|Gamma(U)| out of a maximum, which is valid only when that factor is nonnegative. For |Gamma(U)| >= 3 it is negative and the maximum moves to the minimum of E_P(Y | x). On a twelve-state instance of the source's own recipe with non-constant inference power the accuracy is 0 while the bound is -1/9, so the bound is strictly not attained; Proposition 8 itself still holds there. Recorded as clash 25.", "original_source_refs": [ "survey-ref-054" ], "related_result_ids": [ "CLM-WOLPERT-KNOW-001", "CLM-WOLPERT-EPISTEMIC-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Inference.corollary21_ii_repaired", "AISafetyAtlas.Inference.KnowsEvent", "AISafetyAtlas.Inference.KnowledgeEvent", "AISafetyAtlas.Inference.PositiveIntrospection", "AISafetyAtlas.Examples.Inference.PhysicalKnowledge.Epistemic.corollary21_ii_counterexample", "AISafetyAtlas.Examples.Inference.PhysicalKnowledge.Event.not_positiveIntrospection_observer", "AISafetyAtlas.Inference.prop12_refuted", "AISafetyAtlas.Inference.prop12_gap", "AISafetyAtlas.Inference.prop12_complexity", "AISafetyAtlas.Inference.prop12_entropy", "AISafetyAtlas.Inference.prop12Device", "AISafetyAtlas.Inference.prop12_weaklyInfers", "AISafetyAtlas.Inference.ex6_accuracy", "AISafetyAtlas.Inference.ex6_prop8_bound", "AISafetyAtlas.Inference.ex6_bound_not_attained", "AISafetyAtlas.Inference.ex6_prop8_holds", "AISafetyAtlas.Inference.ex6Device", "AISafetyAtlas.Inference.ex6PMF" ], "module": "AISafetyAtlas.Inference.PhysicalKnowledge.Event", "build_command": "lake build AISafetyAtlas.Inference.PhysicalKnowledge.Event AISafetyAtlas.Examples.Inference.PhysicalKnowledge.Epistemic AISafetyAtlas.Examples.Inference.PhysicalKnowledge.Event" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Inference.corollary21_ii_repaired", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "Limits of self-knowledge and reflection", "summary": "Two tempting epistemic closure claims fail in the source's own device model.", "use": "Executable countermodels prevent false logical-omniscience and positive-introspection claims from entering the API.", "title": "Physical-knowledge boundary cases", "attribution": "Atlas audit of Wolpert 2018" } }, { "id": "CLM-LAWVERE-001", "name": "Lawvere fixed-point theorem (types and functions)", "informal_claim": "If a function f : α → (α → β) is surjective, then every endofunction g : β → β has a fixed point.", "original_source_refs": [ "yanofsky-2003-lawvere" ], "related_result_ids": [ "BY-013", "BY-014", "BY-016", "BY-027" ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/leanprover-community/mathlib4", "version": "db584cd6d46c92f209a44c0f1c829460d327499d", "module": "Mathlib.Logic.Function.Basic", "atlas_module": "AISafetyAtlas.Logic", "declaration": "Function.exists_fixed_point_of_surjective", "relationship": "EQUIVALENT", "reproduced": true, "license": "Apache-2.0", "build_environment": "leanprover/lean4:v4.33.0; mathlib@db584cd6d46c92f209a44c0f1c829460d327499d", "build_command": "lake build AISafetyAtlas.Logic; python3 scripts/check_print_axioms.py" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Logic.lawvere_fixed_point", "type": "WRAPPER", "source_declarations": [ "Function.exists_fixed_point_of_surjective" ] } ] }, "ai_safety_relevance": "A reusable diagonalization kernel. It yields no AI-system limitation without a separate model establishing the required universal representation or surjectivity hypothesis.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Thin wrapper of Mathlib's theorem for ordinary Lean types and functions, graded EQUIVALENT (the word exact is avoided here because EXACT is a distinct grade in this vocabulary). This is the Set/Type specialization of Lawvere's fixed-point argument, not a formalization for arbitrary cartesian closed categories. Its relations to Gödel, Tarski, Löb, and undecidability are mechanism-level only: the existing Atlas theorems retain their syntax, representability, provability, and computability-specific proofs and are not derived from this wrapper.", "tags": [ "computability", "provability-logic" ], "public": { "group": "Limits of self-knowledge and reflection", "summary": "One argument underlies Cantor, Gödel, Tarski and Rice.", "use": "Telling apart arguments that are the same from ones that only rhyme.", "title": "Lawvere fixed point", "attribution": "Lawvere; Yanofsky's uniform treatment" } }, { "id": "CLM-LAWVERE-CCC-001", "name": "Lawvere fixed-point theorem (Cartesian closed categories)", "informal_claim": "In a Cartesian closed category, if there are an object A and a weakly point-surjective morphism A → Y^A, then every endomorphism Y → Y has a fixed point.", "original_source_refs": [ "lawvere-1969-diagonal-ccc" ], "related_result_ids": [ "CLM-LAWVERE-001" ], "formalizations": [], "lean_artifact": null, "ai_safety_relevance": "A general categorical diagonalization schema. It has no AI-system consequence without a model supplying the relevant category, exponential, and weak point-surjectivity assumptions.", "ai_interpretation_status": "HUMAN_REVIEW", "notes": "Source claim recorded from Lawvere's original categorical paper. Atlas currently exposes only the ordinary types-and-functions specialization as CLM-LAWVERE-001; the separate CCC Lean repository discussed in the accompanying audit is unlicensed and is not treated as a reproduced Atlas formalization. A categorical formalization does exist, recorded as NC-005 (2026-08-11): AFP entry Category_Set (The Elementary Theory of the Category of Sets) carries Fixed_Points.thy with lemma Lawveres_fixed_point_theorem for a point-surjective p : X -> A^X over ETCS cfunc/cset with exponential objects, following Halvorson Theorem 2.6.13, and derives Cantor from it. That is a maintained, licensed, pinned categorical rendering. It is not the arbitrary-cartesian-closed-category statement — ETCS is one axiomatized category of sets — so this row still records a source claim rather than a covered one, but the honest reason is generality, not availability. It remains true that Lean supplies no categorical version. A reproduction is feasible with scripts/reproduce_isabelle.sh against the same pinned release and has not been attempted; treat this as a candidate lead, not coverage.", "tags": [ "computability", "provability-logic" ], "statability": { "verdict": "TRIAGED_DISTINCT", "note": "NC-005 (2026-08-11) records a maintained, licensed, pinned categorical rendering: AFP Category_Set carries Lawveres_fixed_point_theorem over ETCS. That is one axiomatized category of sets rather than an arbitrary cartesian closed category, so it does not cover this row's statement; the atlas's own CLM-LAWVERE-001 is the types-and-functions specialization." } }, { "id": "LAND-ATTR-001", "name": "Attribution impossibility (DASH trilemma)", "tags": [ "interpretability" ], "notes": "Core Rashomon/faithfulness trilemma only; upstream GBDT axiom layers not vendored. Not BY-042 or BY-029 coverage without a separate statement map.", "original_source_refs": [], "related_result_ids": [ "BY-029", "BY-042" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/DrakeCaraker/dash-impossibility-lean", "version": "7ec3ef9813a7642fdabe5b73c71d1bed4d5488e2", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; vendored DASH core at upstream Lean v4.29.0-rc8 pin", "module": "DASHImpossibility.Trilemma", "atlas_module": "AISafetyAtlas.Explainability", "build_command": "lake build AISafetyAtlas (vendored axiom-free trilemma)" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Explainability.attribution_impossibility", "type": "WRAPPER", "source_declarations": [ "DASHImpossibility.attribution_impossibility" ], "application": "Explainability: under the Rashomon property no feature ranking is faithful across models within a collinearity group. Reused from DrakeCaraker/dash-impossibility-lean; the atlas exposes the model-agnostic core. Rashomon is witnessed in `Examples.NonVacuity`." }, { "atlas_declaration": "AISafetyAtlas.Explainability.attribution_impossibility_weak", "type": "WRAPPER", "source_declarations": [ "DASHImpossibility.attribution_impossibility_weak" ], "application": "Explainability: weaker packaging of the Rashomon attribution trilemma, requiring less of the ranking. Reused from DrakeCaraker/dash-impossibility-lean alongside the strong form." } ] }, "public": { "group": "Composition and interpretability", "summary": "No way of scoring which inputs mattered meets all the obvious requirements.", "use": "Claims made for saliency maps.", "title": "Attribution impossibility", "attribution": "DASH trilemma" } }, { "id": "LAND-GS-001", "name": "Gibbard–Satterthwaite theorem", "tags": [ "social-choice" ], "notes": "Reproduced AFP landscape formalization (Isabelle). Lean interface is LAND-GS-002. Not a separate Table-1 BY ID; Arrow-session provenance only.", "original_source_refs": [], "related_result_ids": [ "BY-007" ], "root_import": false, "formalizations": [ { "framework": "Isabelle/HOL", "repository": "https://www.isa-afp.org/entries/ArrowImpossibilityGS.html", "version": "AFP release 2026-02-06; archive SHA-256 8174c738b42203100170ff25f3c9fc2c6d16d8556fbaff205c0eaa98a3813da7", "license": "BSD-3-Clause", "reproduced": true, "build_environment": "makarius/isabelle@sha256:9bd33b183c399327c5d554fc8cde27c29b5d2b20cdc6fe7a604caa3f951018fc (Isabelle2025-2; x86_64)", "declaration": "Gibbard_Satterthwaite", "module": "Thys/GS.thy", "build_command": "scripts/reproduce_isabelle.sh arrow" } ], "lean_artifact": null, "public": { "group": "Assembled from other proof assistants", "summary": "The same voting result, proved again in another system.", "use": "Independent confirmation.", "title": "Gibbard–Satterthwaite (independent proof)", "attribution": "Gibbard; Satterthwaite" }, "statability": { "verdict": "EXTERNAL_ONLY", "note": "A formalization exists and is reproduced in Isabelle/HOL: the AFP Gibbard-Satterthwaite development; the atlas Lean interface is the separate row LAND-GS-002. No atlas Lean is owed here -- this row records the external artifact, and reproducing it in Lean would be duplication under lean-parsimony.md rather than coverage." } }, { "id": "LAND-GS-002", "name": "Gibbard–Satterthwaite theorem (Lean / SocialChoiceLean)", "tags": [ "social-choice" ], "notes": "Classical GS: resolute + unanimous + strategy-proof ⇒ dictatorial (3 ≤ |A|). Vendored GS closure from SocialChoiceLean 4.31 port (ballot shim); multi-file mirror under vendor/SocialChoiceLean/. Not full SocialChoiceLean. Isabelle twin is LAND-GS-001. Not a separate Table-1 BY ID; Arrow-session provenance only. Classical resolute voting-rule form.", "original_source_refs": [], "related_result_ids": [ "BY-007" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/DominikPeters/SocialChoiceLean", "version": "mbrcic/SocialChoiceLean port/lean-4.31 @ 74f491b (Mathlib/Lean v4.31.0)", "license": "MIT", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d", "declaration": "gibbard_satterthwaite", "module": "SocialChoice.Impossibilities.GibbardSatterthwaite.Main", "atlas_module": "AISafetyAtlas.SocialChoice", "build_command": "lake build AISafetyAtlas; python3 scripts/check_print_axioms.py" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.SocialChoice.gibbard_satterthwaite", "type": "WRAPPER", "source_declarations": [ "SocialChoice.gibbard_satterthwaite" ] } ] }, "public": { "group": "Aggregation and multi-agent structure", "summary": "Every sensible voting rule can be gamed by lying.", "use": "Asking people or systems to report preferences honestly.", "title": "Gibbard–Satterthwaite", "attribution": "Gibbard; Satterthwaite" } }, { "id": "LAND-NFL-001", "name": "No-free-lunch theorem for machine learning (Shalev-Shwartz–Ben-David §5.1)", "tags": [ "learning-theory" ], "notes": "Statement-level triage (CT-2, 2026-07-19): matches SSBD Thm 5.1 — for binary classification with 2*m < |space X|, every learner admits a realizable distribution on which P[L(A(S)) > 1/8] >= 1/7. Session depends only on HOL-Probability. Reproduction log recorded in docs/provenance/ct2-nfl-triage.md. Lean Wolpert NFL (docs/provenance/lean-wolpert-nfl.md) does not replace or cover this entry. DISTINCT from BY-020 (Wolpert 1996 supervised NFL) and BY-021 (Wolpert–Macready 1997 optimization NFL): this is the PAC / sample-complexity NFL (Understanding ML Thm 5.1), not the uniform-averaging Wolpert results. Not BY-020/021 coverage. See docs/provenance/ct2-nfl-triage.md. The in-tree Lean AISafetyAtlas.Learning.no_free_lunch (BY-021 RELATED) is also DISTINCT from this landscape entry.", "original_source_refs": [ "atlas-ref-shalev-shwartz-ben-david-2014" ], "related_result_ids": [ "BY-020", "BY-021" ], "root_import": false, "formalizations": [ { "framework": "Isabelle/HOL", "repository": "https://www.isa-afp.org/entries/No_Free_Lunch_ML.html", "version": "AFP release 2026-02-06; archive SHA-256 93ce8953bac6b09a29f6d2aafa64d4dbedf49e11f13cdad4cddc42f95f173588", "license": "BSD-3-Clause", "relationship": "RELATED", "reproduced": true, "build_environment": "makarius/isabelle@sha256:9bd33b183c399327c5d554fc8cde27c29b5d2b20cdc6fe7a604caa3f951018fc (Isabelle2025-2; x86_64)", "declaration": "no_free_lunch_ML", "module": "No_Free_Lunch_ML.thy", "build_command": "scripts/reproduce_isabelle.sh nfl" } ], "lean_artifact": null, "public": { "group": "Assembled from other proof assistants", "summary": "A different no-free-lunch theorem from the atlas's: no learner beats a coin flip on every distribution when the sample is small.", "use": "How much data a guarantee needs — not a second proof of the Wolpert averaging results.", "title": "No free lunch (PAC lower bound)", "attribution": "Shalev-Shwartz & Ben-David" }, "statability": { "verdict": "EXTERNAL_ONLY", "note": "A formalization exists and is reproduced in Isabelle/HOL: the AFP PAC / sample-complexity no-free-lunch theorem (Understanding ML Thm 5.1), triaged CT-2 as distinct from BY-020 and BY-021 and from the in-tree Learning.no_free_lunch. No atlas Lean is owed here -- this row records the external artifact, and reproducing it in Lean would be duplication under lean-parsimony.md rather than coverage." } }, { "id": "LAND-PE-001", "name": "Parfit mere addition / conditional normative reasoning (Åqvist E)", "tags": [ "social-choice", "ethics" ], "notes": "A1 (2026-07-19): AFP CondNormReasHOL — Åqvist system E + Parfit mere-addition use case. Not Arrhenius BY-008 coverage. Evidence: docs/provenance/a1-condnorm-parfit-triage.md. RELATED only: population-ethics adjacent (Parfit mere addition / repugnant-conclusion encoding). Not EXACT/EQUIVALENT to Arrhenius 2011 impossibility package (BY-008 survey source). See docs/provenance/a1-condnorm-parfit-triage.md.", "original_source_refs": [ "atlas-ref-parent-benzmueller-condnorm" ], "related_result_ids": [ "BY-008" ], "root_import": false, "formalizations": [ { "framework": "Isabelle/HOL", "repository": "https://www.isa-afp.org/entries/CondNormReasHOL.html", "version": "AFP release 2026-02-06; archive SHA-256 10c3aa794a3cafcfb08a784e11515933162a490b32fc0cdb0cf88f489accdb38", "license": "BSD-3-Clause", "relationship": "RELATED", "reproduced": true, "build_environment": "makarius/isabelle@sha256:9bd33b183c399327c5d554fc8cde27c29b5d2b20cdc6fe7a604caa3f951018fc (Isabelle2025-2; x86_64)", "declaration": "mere_addition", "module": "mere_addition_opt.thy", "build_command": "scripts/reproduce_isabelle.sh condnorm" } ], "lean_artifact": null, "public": { "group": "Assembled from other proof assistants", "summary": "A logic of obligation, precise enough for Parfit's argument.", "use": "Population ethics arguments.", "title": "Conditional normative reasoning", "attribution": "Parent & Benzmüller; Parfit's paradox, Åqvist's system E" }, "statability": { "verdict": "EXTERNAL_ONLY", "note": "A formalization exists and is reproduced in Isabelle/HOL: AFP CondNormReasHOL (Aqvist system E, Parfit mere addition), triaged A1 as RELATED only and not Arrhenius coverage. No atlas Lean is owed here -- this row records the external artifact, and reproducing it in Lean would be duplication under lean-parsimony.md rather than coverage." } }, { "id": "LAND-DL-001", "name": "Deep vs shallow network capacity (Cohen–Bentkamp)", "tags": [ "computational-complexity", "learning-theory" ], "notes": "A2 (2026-07-19): AE almost-all weights, deep model not matchable by shallow width Z < r^(N/2). Landscape RELATED to BY-035; not headline EXACT. Evidence: docs/provenance/a2-deep-learning-by035-triage.md. RELATED to BY-035 no-flattening informal claim (deep nets require large shallow representations). Formalizes Bentkamp/AFP of Cohen et al. 2015-style network-capacity theorems, not the Lin–Tegmark–Rolnick 2017 survey citation as EXACT. See docs/provenance/a2-deep-learning-by035-triage.md.", "original_source_refs": [ "atlas-ref-cohen-sharir-shashua-2016" ], "related_result_ids": [ "BY-035" ], "root_import": false, "formalizations": [ { "framework": "Isabelle/HOL", "repository": "https://www.isa-afp.org/entries/Deep_Learning.html", "version": "AFP release 2026-02-06; full-tree build SHA-256 b059edd46073479ee8dde45004c2346a7365e5d94cded49d27257cfea66c8879; entry archive 018557d0041584239d603a7eb3700d07ed7eb2a2ca48f694820072003ebf430d", "license": "BSD-3-Clause", "relationship": "RELATED", "reproduced": true, "build_environment": "makarius/isabelle@sha256:9bd33b183c399327c5d554fc8cde27c29b5d2b20cdc6fe7a604caa3f951018fc (Isabelle2025-2; x86_64)", "declaration": "fundamental_theorem_network_capacity", "module": "DL_Fundamental_Theorem_Network_Capacity.thy", "build_command": "scripts/reproduce_isabelle.sh deep-learning" } ], "lean_artifact": null, "public": { "group": "Assembled from other proof assistants", "summary": "Deep networks compute things shallow ones need exponential width to match.", "use": "Capability arguments about architecture, not size.", "title": "Deep versus shallow capacity", "attribution": "Cohen, Sharir & Shashua" }, "statability": { "verdict": "EXTERNAL_ONLY", "note": "A formalization exists and is reproduced in Isabelle/HOL: AFP Deep_Learning (Bentkamp, after Cohen et al.), triaged A2 as RELATED to BY-035 and not the Lin-Tegmark-Rolnick citation. No atlas Lean is owed here -- this row records the external artifact, and reproducing it in Lean would be duplication under lean-parsimony.md rather than coverage." } }, { "id": "LAND-VNM-001", "name": "von Neumann–Morgenstern expected utility", "tags": [ "decision-theory" ], "notes": "Reproduced provenance; not an atlas Lake dependency. Adjacent utility foundation; not a BY coverage row.", "original_source_refs": [], "related_result_ids": [], "root_import": false, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/jingyuanli-hk/vNM-Theorem-pub", "version": "89ed1680170bcf947f77bd26cdf614c1ce02222c", "license": "Apache-2.0", "reproduced": true, "build_environment": "Lean 4.31.0; Mathlib fabf563a7c95a166b8d7b6efca11c8b4dc9d911f", "declaration": "vNM.vNM_theorem", "build_command": "scripts/reproduce_vnm.sh" } ], "lean_artifact": null, "public": { "group": "Assembled from other proof assistants", "summary": "Consistent preferences can always be written as a utility function.", "use": "The assumptions inside “maximizes expected utility”.", "title": "Expected utility representation", "attribution": "von Neumann & Morgenstern" }, "statability": { "verdict": "EXTERNAL_ONLY", "note": "A formalization exists and is reproduced in Lean: a reproduced von Neumann-Morgenstern expected-utility development, recorded as provenance and not a Lake dependency. No atlas Lean is owed here -- this row records the external artifact, and reproducing it in Lean would be duplication under lean-parsimony.md rather than coverage." } }, { "id": "LAND-TCS-ARROW-001", "name": "Fourier-analytic Arrow theorem (TCSLib)", "tags": [ "social-choice" ], "notes": "Source-inspected only; deferred until a named theorem needs Fourier structure. Alternative Arrow representation; not additional BY-007 coverage.", "original_source_refs": [], "related_result_ids": [ "BY-007" ], "root_import": false, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/Shilun-Allan-Li/tcslib", "version": "502287d7f8d84c33421c71ce5495b08f097c47a5", "license": "Apache-2.0", "reproduced": false, "declaration": "ArrowTheorem.arrow_theorem" } ], "lean_artifact": null, "statability": { "verdict": "CANDIDATE_LEAD", "note": "TCSLib's Fourier-analytic Arrow theorem is source-inspected only and not reproduced. The row's own note records the reason it stops there: it is an alternative representation of a theorem the atlas already carries, deferred until a named theorem needs Fourier structure. So the lead is live but nothing is blocked on it." } }, { "id": "LAND-DEBATE-001", "name": "Doubly-efficient debate correctness (Brown-Cohen–Irving–Piliouras 2023)", "tags": [ "oversight", "computational-complexity" ], "notes": "CT-7 (2026-07-20): machine-checked correctness of the stochastic-oracle debate protocol, arXiv 2311.14125. Both reproduction lanes are recorded: Path A builds upstream at its own toolchain (Lean/Mathlib v4.8.0) from a separate checkout; Path B (2026-08-20) vendors the LukaHobor Lean v4.31.0 port into AISafetyAtlas/Upstream/Debate/, migrated in place to v4.33.0 on 2026-08-31, where it compiles inside the atlas build closure behind the facade AISafetyAtlas.Oversight.Debate and adds the query-complexity half (alice_fast, bob_fast, vera_fast) that Path A never checked. The port is a toolchain migration only: no theorem weakened, generalized, or removed. The facade is deliberately NOT on the root import - the vendored tree declares roughly 157 root-namespace names - so root_import stays false and the kernel axiom audit reaches it through OFF_ROOT_FACADES in scripts/check_print_axioms.py. The Lipschitz hypothesis is witnessed in-tree by every_oracle_lipschitz_zero, proved here rather than vendored. Upstream caveats carried unchanged: correctness only (space complexity not formalized); time counts oracle queries only; Lipschitz oracle machine defined as a stronger variant. No AI-system reading without a separate reviewed bridge. Evidence: docs/provenance/debate-reproduction.md, vendor/debate/PROVENANCE.md. Possibility / scalable-oversight guarantee, not a Table-1 impossibility row - a live landscape anchor dual to the impossibility rows. No survey coverage.", "original_source_refs": [], "related_result_ids": [], "root_import": false, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/google-deepmind/debate", "version": "de3a6e500ae1a65dfeea2f91ef519ebad9704be0 (main, no release tag; last commit 2024-10-08)", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.8.0; Mathlib v4.8.0 (upstream pinned checkout)", "declarations": [ "completeness", "soundness", "correctness" ], "module": "Debate/Correct.lean", "atlas_module": "AISafetyAtlas.Oversight.Debate", "build_command": "scripts/reproduce_debate.sh" }, { "framework": "Lean", "repository": "https://github.com/google-deepmind/debate", "version": "dafe25df02300c0ebecf436aab32e953006cb0a1 (LukaHobor/debate port-lean-4.31; upstream de3a6e500ae1a65dfeea2f91ef519ebad9704be0)", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; vendored into the atlas build closure", "declarations": [ "completeness", "soundness", "correctness", "alice_fast", "bob_fast", "vera_fast" ], "module": "AISafetyAtlas.Upstream.Debate.Correct", "atlas_module": "AISafetyAtlas.Oversight.Debate", "build_command": "scripts/reproduce_debate.sh --in-tree" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Oversight.Debate.completeness", "type": "WRAPPER", "source_declarations": [ "completeness" ] }, { "atlas_declaration": "AISafetyAtlas.Oversight.Debate.soundness", "type": "WRAPPER", "source_declarations": [ "soundness" ] }, { "atlas_declaration": "AISafetyAtlas.Oversight.Debate.correctness", "type": "WRAPPER", "source_declarations": [ "correctness" ] }, { "atlas_declaration": "AISafetyAtlas.Oversight.Debate.alice_fast", "type": "WRAPPER", "source_declarations": [ "alice_fast" ] }, { "atlas_declaration": "AISafetyAtlas.Oversight.Debate.bob_fast", "type": "WRAPPER", "source_declarations": [ "bob_fast" ] }, { "atlas_declaration": "AISafetyAtlas.Oversight.Debate.vera_fast", "type": "WRAPPER", "source_declarations": [ "vera_fast" ] } ] }, "public": { "group": "Assembled from other proof assistants", "summary": "A weak judge can be argued to the right answer by stronger debaters.", "use": "Judging work you cannot check yourself.", "title": "Doubly-efficient debate", "attribution": "Brown-Cohen, Irving & Piliouras" } }, { "id": "LAND-HYPER-001", "name": "Trace-property and hyperproperty classes (Alpern–Schneider decomposition; Clarkson–Schneider hierarchy)", "tags": [ "compositionality", "verification" ], "notes": "Artifact for Abate, Blanco, Garg, Hritcu, Patrignani and Thibault, 'Journey Beyond Full Abstraction' (CSF 2019, arXiv:1807.04603). Mechanizes the Alpern-Schneider safety/liveness decomposition (safety = closed, liveness = dense, every trace property = safety intersect liveness) and the Clarkson-Schneider class hierarchy including subset-closed hyperproperties, hypersafety, 2-hypersafety, hyperliveness and relational hyperproperties. Found by manual GitHub search on 2026-07-28; it lies outside all six pinned corpora. Exact revision c68187c was tested unmodified. Coq 8.9.1, one of the README's claimed supported versions, and Coq 8.20 both compile Topology.v, TopologyTrace.v, Properties.v and most of the tree, then fail because `Stdlib.List` has no physical load-path binding. Therefore the repository-wide reproduction is FAILED, not reproduced; the principal declarations did kernel-compile before the unrelated late failure. Axiom caveat: Topology.v line 11 declares nonstandard `Axiom prop_ext : prop_extensionality`, used by decomposition_theorem. General arbitrary-k safety and the Clarkson-Schneider self-composition reduction are absent from this upstream tree; the independent axiom-audited Lean increment is LAND-HYPER-002. No Table-1 BY ID concerns hyperproperties; recorded as landscape evidence only.", "original_source_refs": [], "related_result_ids": [], "root_import": false, "formalizations": [ { "framework": "Rocq/Coq", "repository": "https://github.com/secure-compilation/exploring-robust-property-preservation", "version": "c68187cbaba763d88cbb3509df4ca9cf49cc6338", "license": "Apache-2.0", "reproduced": false, "declarations": [ "decomposition_theorem", "safety_closed", "liveness_dense", "decomposition_safety_dense", "Safety", "Liveness", "hprop", "SSC", "HSafe", "H2Safe", "HLiv" ], "modules": [ "Topology.v", "TopologyTrace.v", "Properties.v" ], "build_command": "Exact build attempted 2026-07-28 with coqorg/coq:8.9.1@sha256:8e26609c5450aa795af6917cf169a65ab3952ba666d1b0c90e6b0179314a1149 and coqorg/coq:8.20@sha256:e50d77c4c5a9aa0d76ae1b343d79c5f922da3a75054b79c5dc635895438e4674: both fail at InternalNondet.v line 8 (`From Stdlib Require Import List`) after compiling the principal topology/property files." } ], "lean_artifact": null, "statability": { "verdict": "CANDIDATE_LEAD", "note": "The Coq development for Alpern-Schneider and Clarkson-Schneider was found outside all six pinned corpora and its repository-wide reproduction is recorded FAILED -- the principal declarations kernel-compile, then the tree fails on a Stdlib.List load-path binding. So the artifact is identified and unadjudicated, not absent. The atlas carries its own Compositional.Hyperproperties independently, and since 2026-09-13 that includes a statement of Clarkson and Schneider's Theorem 1 itself -- SubsetClosed, subsetClosed_of_isHyperSafetyOp and a strictness witness, hosted by LAND-HYPER-002. So this row's subject is no longer absent from the tree; what remains unadjudicated is the Coq artifact." } }, { "id": "LAND-HYPER-002", "name": "k-safety self-composition and hyperproperty decomposition", "tags": [ "compositionality", "verification" ], "notes": "Clarkson-Schneider Theorem 2 in both presentations. The batch form uses unordered sets of at most k complete traces. The synchronized-product form k_safety_iff_product_self_composition uses the source ordered Fin k tuples, with the translation and its boundaries made explicit: toBatch is total and coalesces duplicates and order (card_toBatch_le, toBatch_perm), padding recovers a nonempty batch exactly (toBatch_padBatch), and for k greater than zero the empty batch has no product preimage at all, which productSelfComposition_empty and finiteSelfComposition_empty record. The transfer is therefore stated for the induced safety predicate rather than an arbitrary predicate, since selfCompositionSafe_empty_of_any shows the empty batch is subsumed by every other one while an arbitrary predicate could distinguish it; the product form additionally carries a nonempty-system hypothesis. The reduction proves both satisfaction equivalence and that the induced batch predicate has finite bad observations. The operational reading of the decomposition is now earned rather than asserted. PrefixTopology defines the observation topology as a function of the observation relation prefixOf, deliberately not as a global TopologicalSpace instance, since a global instance cannot mention prefixOf and would forget the observation semantics; it is installed locally with letI. Against that topology, isClosed_iff_hyperSafety and dense_iff_hyperLiveness prove that closed and dense mean exactly Clarkson-Schneider hypersafety and hyperliveness, stated independently of any topology, and hyperSafety_of_isKSafety supplies the inclusion joining the k-safety reduction to the decomposition. The generic arbitrary-topology decomposition is retained in the parent module and is no longer the operational claim. Independent Lean proof; it does not import the Rocq artifact or its prop_ext axiom. Non-Table-1 hyperproperty infrastructure; no survey headline coverage. COMPARATOR, 2026-09-13. Bueno and Clarkson, Hyperproperties: Verification of Proofs, Cornell CIS TR, hdl 1813/11153, July 2008, is the only other machine-checked treatment of this theorem; the report and its two Isabelle theory files are held and manifested in the private manifest of 2026-09-13 section 4. It is prior art for a module that already existed, found only because the sources were supplied rather than because the reuse search was re-run. Compared on the printed definitions and NOT machine-checked, since the development does not load in any current Isabelle: it uses 2008-era constdefs, types and axioms commands and its LList2 import was not supplied. Three differences, all in the same direction. FIRST, statement shape: they split the theorem into theorem_2_onlyif and theorem_2_if, each carrying its OWN existential over a safety property, so nothing in either statement pins the two witnesses to the same K; both proofs use Saf_from_KSaf k KK but two separate existentials do not compose to the printed theorem, whereas k_safety_iff_finite_self_composition is a single iff. SECOND, the witness is not quantified away here: SelfCompositionSafe prefixOf k H appears in the statement, where under an existential it does not -- and on their definitions theorem_2_onlyif is discharged by K = TraceInf_K, which is in SPA k because its membership clause is unsatisfiable there and which contains kProd k S by kProd's own definition. THIRD, cardinality: their kProd k S demands a subset of S of cardinality exactly k, where FiniteSelfComposition k S takes batches of at most k. That third difference is where their axiom zip_EX_suffix fails: with k = 2, S a single infinite trace and M two distinct prefixes of it, every hypothesis holds and kProd 2 S is empty, because Systems requires nonemptiness and nothing about cardinality. Their file labels these zip facts as unproved in those words and the report prints the label, so this is a located assumption and not a concealed one. Graded RELATED with the atlas stronger on all three axes. NO formalizations entry, now or ever: the maintainer ruled on 2026-09-13 that this code is used ONLY to check this repository's statements against someone else's and is not, and will not become, part of the atlas. Not vendored, not coverage, not a formalization of anything; it appears in this ledger only as this prose. Eclipse Public License v1.0 therefore stays out of the spdx_license vocabulary and that absence is the decision. Recorded in docs/agent/policy/lean-reuse-sources.md so it is not reopened. THEOREM 3 COMPARED, same source, section 5 of that manifest. Their theorem_3 assumes P in HP where hyperSafety_hyperLiveness_decomposition needs no well-formedness hypothesis, since Hyperproperty is a type here. They have NO topology anywhere: SHP and LHP are the paper's definitions used directly, where this repository proves IsHyperSafetyOp equivalent to IsClosed and IsHyperLivenessOp to Dense in a constructed observation topology and gets the decomposition as a specialization of the generic theorem. And the assumption surfaces are not comparable: their development declares ELEVEN axioms, of which safety_and_liveness_onlyif_true is declared and never used, Cl_produces_Props is admitted in the file to be absent from the accompanying report, and zip_EX_suffix is refutable on their own definitions; their theorem_3 rests transitively on DummyState_is_State through Live_is_hyperliveness and asInfinite_correctness. #print axioms on hyperSafety_hyperLiveness_decomposition returns exactly propext, Classical.choice and Quot.sound, and check_print_axioms enforces that bound on every run. Their theorem_1, SHP strictly inside RC, has NO counterpart here: RC and refinedby have no object in the tree, and that is a named gap rather than an implicit one. THEOREM 1 ADDED 2026-09-13. Clarkson and Schneider's Theorem 1, SHP strictly inside SSC, now has a statement here. SubsetClosed transcribes their SSC from p.1167 of the published paper -- read from a rendered page image -- with their H in HP and T in Prop side conditions structural, since Hyperproperty and TraceSystem are types. subsetClosed_of_isHyperSafetyOp is the inclusion half against the operational hypersafety predicate, and Examples.Compositional.Hyperproperties.subsetClosed_not_hyperSafety is the strictness: a lifted trace property over a liveness trace property, traces the naturals and prefixes the numbers below them, which is subset closed and admits no bad observation because every prefix of a violating trace is also a prefix of a satisfying one. The degenerate route -- an empty Prefix type, where the only observation is the empty one -- was available and deliberately not taken, since it would make strictness an artifact of having nothing to observe. THIS CLOSES A GAP NAMED TWICE UNDER TWO NAMES. Bueno and Clarkson's theorem_1 is SHP strictly inside RC, and their rc quantifies over Systems, which are nonempty, so their RC is this SSC with a nonemptiness restriction; the note recorded on 2026-09-13 that RC and refinedby have no object in this tree described the SAME gap the repository had been carrying as the missing SSC. One gap, two names, now closed once. Their strictness half needs the axiom Ex_nontrue_Prop; this one needs no axiom beyond the three the gate already bounds.", "original_source_refs": [], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "k_safety_iff_finite_self_composition", "k_safety_iff_product_self_composition", "self_composition_is_safety", "isClosed_iff_hyperSafety", "dense_iff_hyperLiveness", "hyperSafety_of_isKSafety", "hyperSafety_hyperLiveness_decomposition", "SubsetClosed", "subsetClosed_of_isHyperSafetyOp" ], "module": "AISafetyAtlas.Compositional.Hyperproperties", "build_command": "lake build AISafetyAtlas.Compositional.Hyperproperties AISafetyAtlas.Compositional.Hyperproperties.PrefixTopology AISafetyAtlas.Compositional.Hyperproperties.Product" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Compositional.Hyperproperties.k_safety_iff_finite_self_composition", "type": "NEW_PROOF", "source_declarations": [ "k_safety_iff_finite_self_composition", "k_safety_iff_product_self_composition", "self_composition_is_safety", "isClosed_iff_hyperSafety", "dense_iff_hyperLiveness", "hyperSafety_of_isKSafety", "hyperSafety_hyperLiveness_decomposition" ], "application": "Compositional verification: k-safety of a system is safety of its k-fold self-composition. Theorem 2 of Clarkson and Schneider (2010)." } ] }, "public": { "group": "Composition and interpretability", "summary": "Some properties need several runs at once to check.", "use": "Leaks and fairness — anything one run cannot reveal.", "title": "Hyperproperties and k-safety", "attribution": "Clarkson & Schneider" } }, { "id": "LAND-ANGLUIN-001", "name": "Port-labelled anonymous networks, views, and automorphisms", "tags": [ "compositionality", "multi-agent" ], "notes": "Builds the network beneath AISafetyAtlas.Compositional.Symmetry, whose Protocol structure takes observational indistinguishability as a field. Here it is derived. runFor_eq_of_view_eq is the Angluin lemma: nodes whose views agree to depth n, meaning every port sequence of length at most n reaches equally-labelled nodes, are in the same state after n synchronous rounds, so an anonymous algorithm cannot see past the depth its messages have travelled. Automorphisms give the second route: invariant_of_automorphism shows a port-preserving permutation leaving the configuration invariant keeps it invariant through every round, and no_unique_leader_of_fixedPointFree concludes that a fixed-point-free automorphism rules out electing a unique leader, since the leader would have to be a fixed point. invariant_of_constant places the earlier constant-configuration symmetry inside this model. step_equivariant is the law underneath the automorphism route, and is stated unconditionally: moving every node by a port-preserving permutation commutes with a synchronous round, with no invariance hypothesis, because anonymity means one update and one send are shared by every node. step_semiconj is that as Function.Semiconj, and invariant_iff_fixed says Invariant is exactly the fixed-point set of the action, so step_invariant is the law read at a fixed point rather than a second induction. These three are infrastructure landed ahead of a consumer by maintainer decision: nothing outside this module uses them yet, and they are prunable through the report_consumers work queue if nothing comes to. The intended consumer is the survey-original BY-043, whose symmetry-breaker taxonomy is stated over exactly this action; no BY-043 status follows from them. Deterministic and synchronous only, so Itai-Rodeh randomized symmetry breaking is out of scope. Message routing is simplified: the message received on port i is what the port-i neighbour sends on index i, rather than being routed back through the neighbour own port, which would need a port involution. The general covering and quotient machinery relating a network to its universal cover is not formalized. Upstream dependency for the survey-original BY-043. Not coverage of it; no status follows from this entry.", "original_source_refs": [], "related_result_ids": [ "BY-043" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "runFor_eq_of_view_eq", "step_equivariant", "step_semiconj", "invariant_iff_fixed", "invariant_of_automorphism", "no_unique_leader_of_fixedPointFree", "invariant_of_constant" ], "module": "AISafetyAtlas.Compositional.Networks", "build_command": "lake build AISafetyAtlas.Compositional.Networks AISafetyAtlas.Examples.SixTargets" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Compositional.Networks.runFor_eq_of_view_eq", "type": "NEW_PROOF", "source_declarations": [ "runFor_eq_of_view_eq", "step_equivariant", "step_semiconj", "invariant_iff_fixed", "invariant_of_automorphism", "no_unique_leader_of_fixedPointFree", "invariant_of_constant" ], "application": "Multi-agent symmetry: nodes with equal views to depth n hold equal states after n synchronous rounds. The Angluin (1980) lemma, over an explicit port-labelled network model." } ] }, "public": { "group": "Aggregation and multi-agent structure", "summary": "Identical processes in identical surroundings can never pick a leader.", "use": "Why identical agents cannot coordinate.", "title": "Symmetry in anonymous networks", "attribution": "Angluin" } }, { "id": "LAND-PREF-KNOW-001", "name": "Reward unidentifiability as a knowability obstruction", "tags": [ "preference-inference", "oversight" ], "notes": "A shared-API formulation, not new coverage. AISafetyAtlas.Preference.policy_neg_twin already states this content in the source's own form - the negated pair explains the same policy - and not_knowable_reward says that in the vocabulary AISafetyAtlas.Knowledge shares with Wireheading and Oversight, so a consumer holding a knowability question can reach it. It is NOT a restatement of the neighbouring policy_reward_unidentifiable, which asserts that every policy/reward pair admits some explaining planner; that is a different statement and the proof of not_knowable_reward does not use it. The world is the planner/reward pair Pair S A, the observation is the evaluated policy op3, the hidden property is Prod.snd, and the collision is the source's own anti-rational negation op4 together with the invariance op3_op4 it already proves. NOT additional BY-011 coverage, and NOT an audit or extension of the source's proof: restating a theorem through Knowable checks nothing about the printed argument. The nonemptiness hypotheses are load-bearing rather than defensive - knowable_reward_of_isEmpty_state proves that with no states the reward type is a subsingleton and recovery succeeds, so a reader who drops [Nonempty S] gets a false statement rather than a failed proof. The statement quantifies over all pairs and therefore excludes nothing about recovery on a restricted class of planners, or of a coarser target such as a reward equivalence class. The module asserts nothing about complexity and does not touch Preference.Complexity's standing non-claim on the anti-rational pair, which it uses only as a collision. Determines.refl and Determines.trans were added to the kernel in the same change and are recorded on LAND-KNOW-001, where they live; what they add there is names rather than laws. Determines is definitionally Knowable - both unfold to (exists f, forall x, B x = f (A x)) - so Determines.trans IS Knowable.mono at the same universes, each provable from the other by direct term application with no tactic, and Determines.refl is the identity-decoder instance of Knowable. Recording them as elementary laws the relation lacked would be wrong: the relation lacked the vocabulary, and Knowable.mono was and remains the single transport. Knowable.mono's own statement writes both names, its first hypothesis as a Determines and its second as a Knowable, which is what made the duplication easy to miss.", "original_source_refs": [], "related_result_ids": [ "BY-011", "LAND-KNOW-001" ], "root_import": true, "relations": [ { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "not_knowable_reward is the kernel's not_knowable_of_collision applied to op3 and Prod.snd; the module proves no factorization law of its own. Determines.refl and Determines.trans are added to the kernel itself rather than here." } ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Preference.not_knowable_reward", "AISafetyAtlas.Preference.knowable_reward_of_isEmpty_state" ], "module": "AISafetyAtlas.Preference.Knowability", "atlas_module": "AISafetyAtlas.Preference", "build_command": "lake build AISafetyAtlas.Preference.Knowability AISafetyAtlas.Examples.Preference.Knowability; python3 scripts/check_print_axioms.py" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Preference.not_knowable_reward", "type": "NEW_PROOF", "source_declarations": [], "application": "Cross-domain: the planner/reward obstruction expressed in the knowability kernel's vocabulary, so Preference is reachable from a question posed about observations. Restatement of existing BY-011 content, not new coverage of it." }, { "atlas_declaration": "AISafetyAtlas.Preference.knowable_reward_of_isEmpty_state", "type": "NEW_PROOF", "source_declarations": [], "application": "The scope boundary, proved rather than asserted: with no states the reward type is a subsingleton and the constant decoder succeeds, so not_knowable_reward's [Nonempty S] cannot be dropped." } ] }, "public": { "group": "What an observer can recover", "summary": "Behaviour alone does not pin down the reward: no single rule turns every policy back into the goal that produced it.", "use": "When behaviour is the only evidence about a goal.", "title": "Reward is not knowable from policy", "attribution": "Armstrong & Mindermann; a shared-API formulation of the row above, no novelty claimed" } }, { "id": "LAND-COMP-TRACE-001", "name": "A network's runs as a trace system", "root_import": true, "original_source_refs": [], "related_result_ids": [ "LAND-ANGLUIN-001", "LAND-HYPER-KNOW-001", "LAND-KNOW-001" ], "tags": [ "compositionality", "oversight" ], "notes": "The first producer of traces in the tree. AISafetyAtlas.Compositional.Hyperproperties defines TraceSystem, Realizes, IsBadObservation, IsKSafety, IsSafetyPredicate, the self-composition reduction and the prefix topology, and nothing anywhere produced a trace; the four modules that generate executions had no consumer reading them as traces, so both halves were live and unconnected. This connects the first of them. A Run is the whole synchronous execution as Nat to Config, a Snapshot is the atom saying that at round n the configuration is c, and realizes_systemOf characterizes exactly which finite observations a network's runs realize: those whose every snapshot is reached at its own round from some admitted initial configuration. NO general answer to what a system is: Trace and Prefix are free type parameters throughout Hyperproperties and prefixOf is supplied by the caller, so this instantiation commits nothing about how any other model should be read, and a shared transition-system interface remains a separate and much larger decision. BOTH READINGS OF A TRACE ARE BUILT. A run can be the round-indexed sequence of whole configurations, which is the outside view a global safety argument wants, or the sequence one node observes, which is the more faithful notion for an anonymous network because it is what a node actually has. ObsTrace is the second, and it is what makes anonymity statable rather than implicit in the absence of identifiers from the Algorithm type: not_knowable_node_of_fixedPointFree takes the node as the unknown, its observation trace as the observation and its own identity as the target, and a fixed-point-free automorphism of an invariant start leaves the observation fixed while moving the identity, which is exactly Knowledge.not_knowable_of_invariant_transform. Nonempty Node is load-bearing there rather than defensive: over an empty node type the decoder is vacuous and the identity is knowable, so dropping the instance gives a false statement rather than a failed proof. no_unique_leader_from_obsTrace is then strictly stronger than the configuration-level result, which evaluates a State to Prop predicate at one round: the predicate may read the node's entire observation history and still cannot single one out, so unbounded memory of a node's own past does not break the symmetry because the symmetry acts on the past too. THE FINER READINGS ARE STILL NOT BUILT: a node's own state sequence is not everything it could see, since Algorithm.update reads incoming messages and the state sequence does not determine them, so a node permitted to record received messages would know strictly more; ObsTrace is the coarsest honest choice rather than the finest possible one. The machinery becomes applicable but the worked example exercises little of it: connecting the types puts IsKSafety, IsHyperSafety, IsHyperLiveness, the self-composition reduction and the prefix topology within reach of a real dynamic model, while what is proved is a single k equals 1 certificate, no k of at least 2 hyperproperty of networks is stated, and non-interference, collusion and privacy under composition remain unformalized over this model. NOT new network mathematics and no BY-043 status follows: not_electsLeader_of_fixedPointFree is no_unique_leader_of_fixedPointFree transported across the producer and proves nothing the Angluin route did not already prove, and Networks remains an upstream dependency of the survey-original BY-043. ElectsLeader is not offered as the interesting hyperproperty; it is the one the tree can already refute, chosen so the producer is exercised by a theorem rather than by a definition. Landed ahead of a consumer by maintainer decision and prunable through the report_consumers work queue. Inherits every non-claim of Compositional.Networks: deterministic and synchronous only, no randomness, no asynchrony, simplified message routing.", "relations": [ { "target": "LAND-ANGLUIN-001", "kind": "BUILDS_ON", "note": "not_electsLeader_of_fixedPointFree is no_unique_leader_of_fixedPointFree transported across the producer; nothing further about networks is proved." }, { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "not_knowable_node_of_fixedPointFree is the kernel's not_knowable_of_invariant_transform at the node-observation map, with the automorphism as the transform; the module proves no factorization law of its own." } ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Compositional.Networks.realizes_systemOf", "AISafetyAtlas.Compositional.Networks.not_electsLeader_of_fixedPointFree", "AISafetyAtlas.Compositional.Networks.electsLeader_witnessed_by_one_snapshot", "AISafetyAtlas.Compositional.Networks.ObsTrace", "AISafetyAtlas.Compositional.Networks.obsTrace_automorphism", "AISafetyAtlas.Compositional.Networks.not_knowable_node_of_fixedPointFree", "AISafetyAtlas.Compositional.Networks.no_unique_leader_from_obsTrace", "AISafetyAtlas.Compositional.Networks.obsSystem", "AISafetyAtlas.Compositional.Networks.stateAtRound", "AISafetyAtlas.Compositional.Networks.realizes_obsSystem", "AISafetyAtlas.Compositional.Networks.systemOf", "AISafetyAtlas.Compositional.Networks.runOf", "AISafetyAtlas.Compositional.Networks.observedAt", "AISafetyAtlas.Compositional.Networks.ElectsLeader" ], "module": "AISafetyAtlas.Compositional.NetworkTraces", "atlas_module": "AISafetyAtlas.Compositional", "build_command": "lake build AISafetyAtlas.Compositional.NetworkTraces; python3 scripts/check_print_axioms.py" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Compositional.Networks.realizes_systemOf", "type": "NEW_PROOF", "source_declarations": [], "application": "The producer/consumer interface: exactly which finite observations a network's runs realize. The trace theory had no producer in the tree before this." }, { "atlas_declaration": "AISafetyAtlas.Compositional.Networks.not_electsLeader_of_fixedPointFree", "type": "NEW_PROOF", "source_declarations": [], "application": "Angluin's conclusion as a statement about a whole trace system rather than one execution at a time. Transported, not re-proved." } ] }, "public": { "group": "Aggregation and multi-agent structure", "summary": "A network's runs can be handed to the machinery that reasons about whole sets of executions.", "use": "When a property is about all runs at once, not one run.", "title": "A network's runs, as a system of traces", "attribution": "after Angluin and Clarkson & Schneider; the connecting construction is the atlas's own, no novelty claimed" } }, { "id": "LAND-HYPER-KNOW-001", "name": "Finite-observation safety as a knowability factorization", "root_import": true, "original_source_refs": [], "related_result_ids": [ "LAND-KNOW-001" ], "tags": [ "compositionality", "oversight" ], "notes": "A shared-API formulation, not new coverage. AISafetyAtlas.Compositional.Hyperproperties.IsSafetyPredicate already says that every violating trace batch is witnessed by one finite observation all of whose realizers also violate; knowable_of_isSafetyPredicate says the same content as a factorization, with the batch as the unknown, the set of finite observations the batch realizes as the observation, and the safety predicate as the target. Two batches realizing the same observations cannot disagree, which is exactly the no-collision condition knowable_iff_no_collision needs. What is added is that the Hyperproperties cluster, which previously had no reference to AISafetyAtlas.Knowledge in either direction and no consumer outside AISafetyAtlas.Compositional, now consumes a shared law, so a question about what an observation settles reaches the trace theory. Landed ahead of a consumer by maintainer decision, and prunable through the report_consumers work queue. NOT a converse: knowability from realizedSet does not give IsSafetyPredicate back, because finiteness of the witnessing observation and the per-violation witness are both lost on the way in, and no attempt is made to recover them. NOT a trace producer: this connects the consumer side of the trace theory to the kernel and nothing here produces a trace. The producer is AISafetyAtlas.Compositional.NetworkTraces, catalogued as LAND-COMP-TRACE-001, which reads one of the four execution-generating modules as a TraceSystem; the other three, Compositional.Symmetry, Wireheading.CRMDP and Wireheading.GoalPreservation, still have no trace consumer. NOT a statement about IsKSafety, IsHyperSafety or IsHyperLiveness; only the ordinary batch predicate is routed. The property lands in Prop, so knowable_iff_no_collision's [Nonempty Y] is discharged by Prop being inhabited and the collision step closes with propext. Nothing here needs prefixOf decidable or Prefix and Trace finite.", "relations": [ { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "knowable_of_isSafetyPredicate is the kernel's knowable_iff_no_collision applied to the realized-observation map and the safety predicate; the module proves no factorization law of its own." } ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Compositional.Hyperproperties.knowable_of_isSafetyPredicate", "AISafetyAtlas.Compositional.Hyperproperties.not_knowable_of_realizedSet_collision", "AISafetyAtlas.Compositional.Hyperproperties.not_isSafetyPredicate_of_realizedSet_collision", "AISafetyAtlas.Compositional.Hyperproperties.realizedSet" ], "module": "AISafetyAtlas.Compositional.Hyperproperties.Knowability", "atlas_module": "AISafetyAtlas.Compositional", "build_command": "lake build AISafetyAtlas.Compositional.Hyperproperties.Knowability; python3 scripts/check_print_axioms.py" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Compositional.Hyperproperties.knowable_of_isSafetyPredicate", "type": "NEW_PROOF", "source_declarations": [], "application": "Cross-domain: finite-observation safety expressed in the knowability kernel's vocabulary, so a question about what an observation settles reaches the trace theory. Restatement of existing Hyperproperties content, not new coverage." }, { "atlas_declaration": "AISafetyAtlas.Compositional.Hyperproperties.not_isSafetyPredicate_of_realizedSet_collision", "type": "NEW_PROOF", "source_declarations": [], "application": "The certificate direction a consumer holding a counterexample has: two batches realizing the same observations but disagreeing refute safety outright." } ] }, "public": { "group": "What an observer can recover", "summary": "If a safety check reads only finitely much of a run, two runs that look alike must get the same verdict.", "use": "When a monitor sees only finite evidence.", "title": "Finite evidence decides a safety verdict", "attribution": "after Clarkson & Schneider; a shared-API formulation, no novelty claimed" } }, { "id": "LAND-CAUSAL-KNOW-001", "name": "Behavioural identifiability as a knowability factorization", "root_import": true, "original_source_refs": [], "related_result_ids": [ "LAND-CAUSAL-DECISION-001", "LAND-KNOW-001" ], "tags": [ "interpretability", "agent-incentives", "oversight" ], "notes": "A shared-API formulation, not new coverage and no MAIS status follows from it. AISafetyAtlas.Causal.Skeleton.BehaviorEq already says two models agree on the masked gap response over every admissible visible set, assignment and mixture; behaviorEq_iff_behavior_eq proves that this is equality of a single observation map, in both directions, so bundling the bounded quantifier is a faithful repackaging rather than a change of statement. not_knowable_of_behaviorEq then reads a behaviourally equal pair with a differing attribute as a Knowledge collision. The generic form is what is new here, not the phenomenon: AISafetyAtlas.Examples.Causal already builds such pairs by hand in margin_class_not_identifiable, margin_class_not_identifiable_two_graphs, margin_class_not_identifiable_family and margin_class_not_identifiable_shared_optimal, and in the AISafetyAtlas.Examples.Causal.BehavioralCollision construction that answers MAIS-O23 in the negative. What those do not supply is the passage from a negated knowability to such a pair, which is exists_behaviorEq_pair_of_not_knowable, the kernel's exists_witness_of_not_knowable relabelled and classical as it is there. The Causal cluster previously had no reference to AISafetyAtlas.Knowledge in either direction, so this is the cluster's first shared-law edge. Landed ahead of a consumer by maintainer decision, and prunable through the report_consumers work queue. NOT a statement about InIdentifiedSet: that relation is about admissible policy families at a regret bound, a strictly weaker separator than behavioural equality, and inIdentifiedSet_zero_of_behaviorEq runs one way with no converse proved in the cluster; nothing here claims the identified set is the collision set of behavior. NOT a claim that any attribute is unidentifiable: every result is conditional on a supplied pair or a supplied failure. NOT an audit: restating identifiability through Knowable checks nothing about the printed argument. The observation is the masked gap family and nothing else, so a consumer whose policies read more than Delta-mask is outside every statement here.", "relations": [ { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "Every theorem here is a kernel law at the causal observation: the collision law, the classical witness extraction, and Knowable.mono. The module proves no factorization law of its own." }, { "target": "LAND-CAUSAL-DECISION-001", "kind": "BUILDS_ON", "note": "Reads Skeleton, Model.Delta-mask and BehaviorEq from the decision layer and restates them; proves nothing further about causal models." } ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Causal.behaviorEq_iff_behavior_eq", "AISafetyAtlas.Causal.not_knowable_of_behaviorEq", "AISafetyAtlas.Causal.exists_behaviorEq_pair_of_not_knowable", "AISafetyAtlas.Causal.knowable_of_determines_behavior", "AISafetyAtlas.Causal.behavior" ], "module": "AISafetyAtlas.Causal.Knowability", "atlas_module": "AISafetyAtlas.Causal.Knowability", "build_command": "lake build AISafetyAtlas.Causal.Knowability; python3 scripts/check_print_axioms.py" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Causal.behaviorEq_iff_behavior_eq", "type": "NEW_PROOF", "source_declarations": [], "application": "The faithfulness of the repackaging, proved in both directions: BehaviorEq's bounded quantifier bundled as a dependent function is the same statement." }, { "atlas_declaration": "AISafetyAtlas.Causal.exists_behaviorEq_pair_of_not_knowable", "type": "NEW_PROOF", "source_declarations": [], "application": "Cross-domain: from a failure of identifiability to a concrete behaviourally equal pair. The direction the hand-built collisions in Examples.Causal do not supply." } ] }, "public": { "group": "What an observer can recover", "summary": "If two causal models act the same under every test, nothing that tells them apart can be read off their behaviour.", "use": "When behaviour is the only evidence about a causal structure.", "title": "Behaviour does not pin down a causal model", "attribution": "a shared-API formulation over the atlas's own causal decision layer, no novelty claimed" } }, { "id": "LAND-COMP-KNOW-001", "name": "The Angluin view as a knowability factorization", "tags": [ "compositionality", "oversight" ], "notes": "A shared-API formulation, not new coverage. AISafetyAtlas.Compositional.Networks.runFor_eq_of_view_eq already proves the Angluin lemma - nodes whose views agree to depth n are in the same state after n rounds - and knowable_runFor says the same thing as a factorization: the node is the unknown, the depth-n view is the observation, the state after n rounds is the target property, and the Angluin lemma is exactly the no-collision condition knowable_iff_no_collision needs. NOT new BY-043 coverage: Networks is an upstream dependency for the survey-original result and remains one, and no status upgrade follows. NOT a new theorem about networks: the mathematics is entirely runFor_eq_of_view_eq's. What is added is that Compositional, which previously had no cross-domain library import in either direction, now consumes a shared law. It concludes Knowable rather than refuting it, which the kernel also does in Knowledge.Devices.knowable_probe_of_forall_blockAnswers, Knowledge.Check.knowable_of_findCollision_eq_none, and twice in Oversight.JointObservation. knowable_runFor carries no hypotheses, because the Angluin lemma holds for every network, algorithm, configuration and round count without qualification. No uniqueness claim is attached to it. Restating an existing theorem in a shared vocabulary is worth doing whether or not the restatement is the only one of its shape. The repackaging of SameView as a function on port sequences of bounded length is proved faithful in both directions by sameView_iff_view_eq, so nothing is lost or added on the way into the kernel. The statement is unconditional in the state type: knowable_iff_no_collision's [Nonempty Y] is discharged by a case split rather than imposed, because with no states there are no nodes either and the observation type is itself empty. Inherits every non-claim of Compositional.Networks - deterministic and synchronous only, no randomness, no asynchrony, simplified message routing - and asserts nothing about whether a node can compute the decoder, which is assembled classically. The cost is measured rather than asserted: runFor_eq_of_view_eq depends on propext and Quot.sound, and this proof of knowable_runFor adds Classical.choice, because it assembles the decoder through Function.extend, which needs a junk value on views no node realizes. That is a fact about this implementation and not about the theorem - nothing here shows the factorization requires choice, and no claim of strict logical strength is made either way.", "original_source_refs": [], "related_result_ids": [ "LAND-ANGLUIN-001", "LAND-KNOW-001" ], "root_import": true, "relations": [ { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "knowable_runFor is the kernel's knowable_iff_no_collision applied to the depth-n view and the state after n rounds; the module proves no factorization law of its own." }, { "target": "LAND-ANGLUIN-001", "kind": "BUILDS_ON", "note": "The no-collision premise is runFor_eq_of_view_eq verbatim. This row restates that theorem and proves nothing further about networks." } ], "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Compositional.Networks.knowable_runFor", "AISafetyAtlas.Compositional.Networks.sameView_iff_view_eq", "AISafetyAtlas.Compositional.Networks.view" ], "module": "AISafetyAtlas.Compositional.Knowability", "atlas_module": "AISafetyAtlas.Compositional", "build_command": "lake build AISafetyAtlas.Compositional.Knowability AISafetyAtlas.Examples.Compositional.Knowability; python3 scripts/check_print_axioms.py" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Compositional.Networks.knowable_runFor", "type": "NEW_PROOF", "source_declarations": [], "application": "Cross-domain: the Angluin lemma expressed in the knowability kernel's vocabulary, so a question about what an observation settles reaches the network model. Restatement of existing LAND-ANGLUIN-001 content, not new coverage of it." }, { "atlas_declaration": "AISafetyAtlas.Compositional.Networks.sameView_iff_view_eq", "type": "NEW_PROOF", "source_declarations": [], "application": "The faithfulness of the repackaging, proved in both directions: an observation map must return one value, and SameView is a quantified family of equalities." } ] }, "public": { "group": "What an observer can recover", "summary": "In an anonymous network, how far a node can see decides what state it ends up in.", "use": "The positive side of the same law: when a limited observation is nonetheless enough.", "title": "A node's view decides its state", "attribution": "after Angluin; a shared-API formulation, no novelty claimed" } }, { "id": "LAND-RECT-001", "name": "Rectangle, exchange, and unary-contract equivalences", "tags": [ "compositionality" ], "notes": "Mechanizes the communication-complexity mix-and-match characterization for binary relations and two indexed full-product characterizations. Sources: Kushilevitz-Nisan (1997) rectangle folklore and Fagin (1977) lossless-join/MVD special case. coordinate_product_iff_spliceClosed is the substantive indexed result: over a finite index and a nonempty predicate, being the full product of unary contracts is equivalent to closure under updating one accepted point at a single coordinate, proved by splicing coordinates one at a time along a Finset.piecewise induction. coordinate_product_iff_recombination_closed is retained only as a helper and is explicitly demoted in the module docstring, since one inclusion of the product equality always holds and the statement therefore reduces to unfolding Set.pi. Necessity of finiteness is machine-checked rather than asserted: spliceClosed_finitelySupported and not_isCoordinateProduct_finitelySupported show the finite-support Boolean sequences over the naturals are splice closed with full unary projections yet omit the constantly-true sequence. Applies to single trace/state predicates, not hyperproperties. Compositional trace/state mathematics, not a Table-1 impossibility result.", "original_source_refs": [], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "rectangle_iff_exchange_closed", "coordinate_product_iff_spliceClosed", "not_isCoordinateProduct_finitelySupported", "coordinate_product_iff_recombination_closed" ], "module": "AISafetyAtlas.Compositional.Rectangularity", "build_command": "lake build AISafetyAtlas.Compositional.Rectangularity" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Compositional.rectangle_iff_exchange_closed", "type": "NEW_PROOF", "source_declarations": [ "rectangle_iff_exchange_closed", "coordinate_product_iff_spliceClosed", "not_isCoordinateProduct_finitelySupported", "coordinate_product_iff_recombination_closed" ], "application": "Compositional boundaries: a binary relation is a combinatorial rectangle exactly when it is exchange-closed. Kushilevitz-Nisan (1997) rectangle characterization." } ] }, "public": { "group": "Composition and interpretability", "summary": "A requirement on the whole sometimes splits into requirements on the parts.", "use": "When checking components separately is enough.", "title": "Rectangularity", "attribution": "Kushilevitz & Nisan; Fagin" } }, { "id": "LAND-WIRE-OBJ-001", "name": "Ring-Orseau objective factorization", "tags": [ "agent-incentives" ], "notes": "The Ring and Orseau (AGI 2011) agent equations and their finite-horizon consequences. AgentEquations writes down the source displayed equations (1) to (3): value is the depth-truncated form of the mutually recursive v_t(h) and v_t(ha), bestAction and bestAction_max are the argmax of equation (1), and actionValue is equation (2). Equations (2) and (3) have no base case in the source, so the truncation is explicit rather than implicit: value_eq_zero_of_horizon_vanishes and truncation_exact show that once the horizon weighting vanishes past the window, deepening the recursion changes nothing, which is the condition under which the finite form is the source value rather than an approximation. AgentEquations.value_eq_of_agree_on_window is the factorization with content, since its proof unfolds the recursion: with the prior fixed, the depth-n value depends only on the utility and horizon inside the reachable window. The Objective module keeps the same shape at one further remove, with its own locality, homogeneity and rescaling results, and separately. The headline results unfold the horizon sum: value_eq_of_agree_on_window is the truncation statement that value on a window depends only on the weights and utilities inside it, so two objectives may diverge arbitrarily outside and still agree on it; value_scaleUtility is homogeneity in utility; optimal_decisions_eq_of_pos_scaleUtility carries that through to optimal-decision sets under strictly positive rescaling. Separately, value_congr and optimal_decisions_congr are record congruence only: their proofs conclude equality of the two objectives from componentwise equality and never unfold value, so they are well-definedness lemmas and not results about objectives. An earlier revision exposed only those two under the name objective_factorization, which overstated their content; they are renamed and demoted here. The paper frames its results as statements and arguments and supplies no numbered objective-factorization theorem, so nothing here reproduces a numbered source result. Lean utilities are unrestricted real values rather than source-bounded [0,1] values. DelusionBox adds the paper section 3 construction and the three statements argued about it. GlobalEnv is the global environment split into an inner environment and a box whose program is carried inside the agent's action, with the paper's two prose assumptions as theorems rather than restatements: innerHistory_congr_innerAction says the inner environment cannot read the program, and globalObs_indep_inner says a constant program makes two global environments indistinguishable that an untouched box tells apart. statement_one, statement_two and statement_three are the paper's Statements 1 to 3 at its own thresholds, and bestAction_ne_of_lt carries each to equation (1). THEY WERE GRADED Partial AND Narrower UNTIL 2026-09-20 AND ARE NOW Yes AND Wider in section 11 of docs/provenance/source-coverage-audit.md. The paper argues rather than proves them, and its arguments need premises it never states: a two-hypothesis decomposition of a value and four bounds on branch values. Those are still named explicit hypotheses of the Statement theorems rather than strengthenings of the agent definitions - nothing is folded into rlAgent - and every one of them is now DISCHARGED from the paper's own construction. THE DECOMPOSITION WAS REFUTED, 2026-09-19: Examples not_mixesAt_one shows a fixed-weight average is not a mixture's value above remaining depth zero, and AISafetyAtlas.Wireheading.Mixture replaces it with a posterior and a committed policy, whose value mixes exactly because no maximum intervenes. mixesAt_zero survives as the depth-zero case. THE FOUR BRANCH BOUNDS AND r-bar < 1 WENT 2026-09-20. rlAgent_actionValue_next_const is the paper's 'the agent can program the DB to produce a constant reward of 1' - at a GlobalEnv whose box runs a program constant at a reward-one observation the value is that reward EXACTLY, by GlobalEnv.next_const, where print writes an inequality. rlAgent_actionValue_nonneg and rlAgent_actionValue_le are the >= 0 and <= 1 bounds from a bounded reward and Belief.IsSubprobability, which are print's own u into [0,1] and its rho a probability; diracBelief_isSubprobability supplies the latter for the optimal non-learning variant. The fourth bound is PRINT'S DEFINITION of r-bar, 'the expected reward when not using the DB', and rlAgent_actionValue_zero is that expectation, so nothing derives it and nothing should. threshold_unattainable_of_one_le shows print's threshold exceeds every probability for 1 <= r-bar < 2, so the Statement is vacuous there, and not_mixture_lt_of_threshold_without_rbar shows the comparison is FALSE above 2 - so r-bar < 1 is forced by print's own threshold rather than added. ALL FOUR DERIVATIONS ARE AT REMAINING DEPTH ZERO AND THAT IS PRINT'S READING: print's bounds are rewards and horizon weights, not sums of them, and at depth n the same construction is worth n+1 rather than 1. STATEMENT 2'S CONSTANTS ARE THE HORIZON, NOT THE PRIOR, and the audit's cost note said otherwise. 2^-j is shortHorizon t (t+j), the goal-seeking discount for reaching the goal j steps later; the universal prior enters print's Statement 2 paragraph nowhere. shortHorizon_ahead is the computation, goalAgent_actionValue_next_const is print's v(h yes) bound as an equality, and goalAgent_actionValue_zero_of_not_goal is the refusing action's value where the goal is not reachable at the next step, which is the l-a = infinity case and NOT print's v(h no) bound. STATEMENT 2 NEEDED A THIRD AXIS THAT THE CLOSING OF STATEMENTS 1 AND 3 EXPOSED, AND IT CLOSED THE SAME DAY. Print's 2^-l-a and 2^-|o+| are horizon weights of steps LATER, so unlike Statement 1's rewards they are not one-step quantities and both bounds have to hold at every depth. ReachedAtMostOnce is print's own side condition on the goal predicate, stated on its page as 'the goal can be reached at most once, so sum_t u(h_t) <= 1' and recorded in section 11 since 2026-09-09 as a condition NOTHING USED; goalAgent_value_le_shortHorizon is what it was stated for, and without it the goal-seeking value exceeds the horizon weight of its own step by a factor of two because the weights 2^(t-k) sum to twice the first. GoalOutOfReach is print's l-a RELATIVE TO THE BELIEF, since print's l-a exceeds |o+| because the environment is slower rather than because the goal is longer, and goalAgent_actionValue_le_of_outOfReach is print's v(h no) bound at every depth. ReachesIn and shortHorizon_le_goalAgent_policyActionValue are print's v(h yes) bound at every |o+|, through a COMMITTED POLICY - the same object Statement 1's repair needed, because equation (3)'s maximum gives no lower bound. Examples goal_gap is print's 'easily satisfiable once |o+| < l-a' on a real pair of environments, and on print's own case rather than a degenerate one: the box reaches the goal one step on and the slow environment reaches it TWO steps on, so l-a = 2 > 1 = |o+| and the bounds are 2^-1 against 2^-2. Examples slow_reachesIn_two shows the goal is genuinely reached without the box, which is what print's l-a >= |o+| asserts and what an unreachable goal would not exhibit. The depth-zero lemma goalAgent_actionValue_zero_of_not_goal remains as the never-reachable case; goalAgent_actionValue_le_of_outOfReach is the one print's argument needs. STATEMENT 3'S CONVERGENCE AXIS IS RETRACTED, NOT PAID. Print's Arguments run on an inequality between the true environment's weight and the mass of the class containing it - now ProgramPrior.Model.setMass, weight_le_setMass and the strict weight_lt_setMass - and then on 'it takes fewer errors to converge' to the class, which print's preceding sentence sources to another paper: 'a predictor makes approximately -log(rho(q)) errors [2]'. That is a cited theorem of [2] and not a claim of this paper, so the atlas is not owed it. Statement 4 is graded Partial: historyMass and coherentKnowledgeAgent restore print's u(h) = -rho(h) at the observation layer, so the statement is instantiable at print's agent; the conclusion is not proved. knowledge_uses_the_box still refutes it at the untied knowledgeAgent; coherent_uses_the_box shows the coherent agent at the mixture belief still programs the box. Print's argument about discarded program mass needs a type of programs with a prior over it. SelfMod, added 2026-09-11, is the companion paper's section 3 substrate: Exec is the executor, smValue is equation (4) at finite depth (the continuation is valued at the next code, not by maximising again), and survivalAgent is A_s. The six Statements and the Simpleton Gambit remain No. The companion paper Orseau and Ring, Self-Modification and Mortality in Artificial Agents, is graded in section 13 of the source coverage audit as of 2026-09-10; its section 2 is this paper's section 2, so several declarations in this row cover it too, and it prints DIFFERENT horizon functions for the goal-seeking and knowledge-seeking agents, which is why those two graded Partial there and Yes here. THE GOAL-SEEKING HALF OF THAT CLOSED 2026-09-13: page 4 of the mortality paper (sha256 e211682a..., read as a rendered image) says 'The goal can be reached at most once, so sum_t u(h_t) <= 1. For this utility function, the horizon function is not necessary, does not need to be summable, and can be set to 1: w(t,k) = 1.' companionGoalAgent is that agent, at the new unitHorizon, with companionGoalAgent_utility_eq recording that it shares goalAgent's utility and goalAgent_horizon_ne that the horizons differ at t=0, k=1, so neither record is a proxy for the other paper's agent. Section 13's A_g row went Partial/Narrower to Yes/Same. The knowledge-seeking half does NOT close the same way: the two papers print k-t=m and k+t=m and the audit takes no view on which is meant, so that row stays Mixed by decision rather than by omission. OBSERVATION-TYPE AXIS CLOSED 2026-09-13: equation (2)'s sum was a Finset sum over a Fintype Obs and is now an unconditional sum over an arbitrary observation type, because print's setup page bounds neither A nor O - it fixes only a in A and o in O. No summability hypothesis was added to buy it: value_eq_of_agree_on_window, value_eq_zero_of_horizon_vanishes and truncation_exact need only congruence and the vanishing case, both of which hold of the unconditional sum, and actionValue_eq_sum recovers the finite sum at a Fintype so the old signature is an instance of the new one. SelfMod's smValue was freed in the same way and now carries NO finiteness at all - it values the continuation at the next code rather than by maximising again, so it never needed the action instances either. mixesAt_zero still takes [Fintype Obs] explicitly and is the only declaration in the cluster that does, because it splits a sum of two families and that is false without summability. THE TWO IMPLICATIONS THE WIDENINGS RESTED ON ARE NOW PROVED, 2026-09-13. The unconditional sum and the conditional supremum are both zero by convention off their domains, and until today the claim that at print's hypotheses they denote print's value was a statement this tree did not carry. AgentEquations.actionValue_summable and AgentEquations.actionValue_bddAbove carry it, at Belief.IsSubprobability and bounded utility, which are print's own rho-a-probability and u : H -> [0,1]. actionValue_eq_sum and value_succ_eq_sup' still recover the finite cases. WHAT SURVIVES: bounded is not attained, so Attains remains a hypothesis and is not derived from the bound, and where print's maximum is not attained the supremum and the maximum still differ. NEW MODULE ValueBounds supplies the infinite-horizon limit the printed equations never define: infiniteValue, tendsto_value_infiniteValue, the certified truncation error value_error_le_tail, abs_infiniteValue_le, and infiniteValue_eq, which is equation (3) at that limit. Its hypotheses are the mortality paper's page 2 verbatim - rho a positive prior whose total mass must be finite, u into [0,1], and 'in general, it must be summable: sum_{k=t}^inf w(t,k) < inf'. The delusion paper prints no summability clause, so section 11's equation (3) row names the condition and attributes it to the companion. NEW MODULE ProgramPrior is the program layer four audit rows were costed for: Consistent is print's Q_h, mass is its rho(h) := sum over Q_h, mass_eq_historyMass joins it to the observation layer, belief_isSubprobability feeds the value bounds, point is the optimal non-learning variant's concentration on a program with point_belief_eq and value_point_eq showing it induces DelusionBox.diracBelief and the same values on-support, knowledgeAgent is print's u(h) = -rho(h) at the agent's OWN mass, and relative_weight_le is print's page-4 monotonicity. Print's rho is strictly positive on Q and the atlas's weight is only nonnegative, which is wider and is what admits point. STATEMENT 4 IS STILL NOT PROVED: the layer its argument needs now exists, nothing here exhibits print's own agent programming the box, and nothing here proves it does not. Six audit cells moved: sections 11 and 13's equation rows to Wider, section 13's A_k to Yes and Wider, and section 11's optimal variants and Statement 4 and section 13's A^mu to Same, each staying Partial for a reason the row names. ACTION-TYPE AXIS CLOSED 2026-09-13, by the route the costing named: value's recursion maximised over actions with Finset.sup', which needs the maximum attained, and now takes the conditional supremum over an arbitrary action type, while bestAction takes the hypothesis Attains - strictly weaker than a Fintype, and exactly what print presupposes by writing an argmax. attains_of_fintype and value_succ_eq_sup' recover the finite readings, so the previous signatures are instances of the new ones. No hypothesis was added to buy it: value, actionValue, value_eq_of_agree_on_window, value_eq_zero_of_horizon_vanishes and truncation_exact need attainment nowhere, and four of them turned out not to need Nonempty Action either, so the module and DelusionBox now carry no action instances at all. DelusionBox.bestAction_ne_of_lt gained the Attains argument and the four Examples call sites discharge it with attains_of_fintype. The audit's equation (1) rows in sections 11 and 13 went Narrower to Same. WHAT THIS WIDENING DOES NOT PROVE: the conditional supremum is zero by convention at a family that is empty or unbounded above, and at print's own hypotheses (u into [0,1], w summable) the family is bounded and the supremum is print's maximum - that implication is not formalized in this tree, and value_succ_eq_sup' covers the finite case only. Where the maximum is not attained this module states a supremum and print states a maximum, and those differ. This entry does not formalize AIXI, Solomonoff induction, the paper's convergence route to its thresholds, its section 4 self-modification setting, or asymptotic convergence. Supporting wireheading mathematics, not itself a Table-1 row. THE INFINITE RECURSION OF EQUATION (4) NOW HAS A LIMIT, 2026-09-14, AND THAT IS ALL IT HAS. SelfMod.abs_smValue_le_horizonBudget bounds the truncated self-modifying value at a subprobability belief and a bounded utility, abs_smValue_succ_sub_le bounds the step, SelfMod.smInfiniteValue is the limit, tendsto_smValue_smInfiniteValue is the convergence and smValue_error_le_tail the certified truncation error - the same shape ValueBounds gives the companion's equation (3), now for the recursion that values the continuation at the NEXT code. Examples.Wireheading.SelfMod.smInfiniteValue_stays inhabits it. WHAT IT DOES NOT SUPPLY, and section 13's equation (4) row stays Mixed for exactly this: a convergent value is not a chosen initial program. The argmax over compound actions, the attainment of that maximum, and the realization of the maximizing program are three separate obligations and none of them is here. ProgramPrior.expectation_eq_programSum is the second addition: for any uniformly bounded continuation the observation expectation equals the consistent-program sum divided by the history mass, with `actionValue_eq_programSum` the same statement at the action value and Examples.Wireheading.ProgramPrior.posterior_true the witness. It is the posterior formula Statements 1 to 3 argue over; it is NOT those statements, which stay No, and nothing here supplies the concrete program classes or the posterior thresholds their arguments need.", "original_source_refs": [], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AgentEquations.value_eq_of_agree_on_window", "AgentEquations.truncation_exact", "AgentEquations.bestAction_max", "Objective.value_eq_of_agree_on_window", "Objective.value_scaleUtility", "Objective.optimal_decisions_eq_of_pos_scaleUtility", "Objective.value_congr", "Objective.optimal_decisions_congr", "DelusionBox.GlobalEnv.innerHistory_congr_innerAction", "DelusionBox.GlobalEnv.globalObs_identity", "DelusionBox.GlobalEnv.globalObs_const_of_isConstant", "DelusionBox.GlobalEnv.globalObs_indep_inner", "DelusionBox.actionValue_diracBelief", "DelusionBox.mixesAt_zero", "DelusionBox.statement_one", "DelusionBox.statement_two", "DelusionBox.statement_three", "DelusionBox.bestAction_ne_of_lt", "SelfMod.Exec", "SelfMod.smValue", "SelfMod.survivalAgent" ], "module": "AISafetyAtlas.Wireheading.Objective", "build_command": "lake build AISafetyAtlas.Wireheading.Objective AISafetyAtlas.Wireheading.AgentEquations AISafetyAtlas.Wireheading.DelusionBox AISafetyAtlas.Wireheading.SelfMod AISafetyAtlas.Examples.Wireheading.DelusionBox AISafetyAtlas.Examples.Wireheading.SelfMod" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Wireheading.AgentEquations.value_eq_of_agree_on_window", "type": "NEW_PROOF", "source_declarations": [ "AgentEquations.value_eq_of_agree_on_window", "AgentEquations.truncation_exact", "AgentEquations.bestAction_max", "Objective.value_eq_of_agree_on_window", "Objective.value_scaleUtility", "Objective.optimal_decisions_eq_of_pos_scaleUtility", "Objective.value_congr", "Objective.optimal_decisions_congr" ], "application": "Self-modification: an agent's finite-horizon value depends only on the percept window it can still act on. Agent equations of Ring and Orseau (AGI 2011)." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "An agent can end up caring about what it sees, not about what is real.", "use": "Exactly when wireheading is optimal.", "title": "Objective factorization", "attribution": "Ring & Orseau" } }, { "id": "LAND-WIRE-AGENTHISTORY-001", "name": "Ring-Orseau histories on the Decision carrier", "tags": [ "agent-incentives" ], "notes": "A JOIN, NOT A NEW RESULT. AgentEquations evaluates value by recursion along its own interaction history, List (Action x Obs), and the Decision cluster runs policies along Obs x List (Action x Obs). Nothing related the two, so a value recursion could not be evaluated along a run and a run could not be scored by an agent's utility. AgentHistory is the transport: toDecisionHistory pairs an interaction history with an initial observation, ofDecisionHistory forgets it, and BOTH ROUND TRIPS ARE rfl - the two carriers are the same list with one datum in front, which is why this is a join and not a translation. WHAT THE JOIN IS FOR. runHistory is the interaction history a determined run produces, runHistory_length is that a run of n steps produces a history of length n, and value_runHistory_eq_of_agree_on_window is AgentEquations.value_eq_of_agree_on_window stated where the history is one a policy and a transition actually produced and the window is counted in STEPS. Stated at a bare list the factorization is a claim about lists; stated along a run it is a claim about an agent in a world. WHY IT WAS BUILT. Two private elicitation notes (2026-08) name seven incompatible notions of agent behaviour in this tree with no transports between any of them, and call that a present-tense defect: interoperable, not silos. Wireheading.CRMDP already sits on Decision.MDP; this is the second of the seven to reach the shared carrier. WHAT IT DOES NOT CLAIM. No reward and no corruption - the transport carries histories, and CRMDP is the module that puts a reward on this carrier. Determined runs only: the drawn run is a PMF and relating value to an expectation over it is a different statement, not proved here. Agent and MDP remain distinct structures. THE WITNESS IS THE POINT, AND IT REPLACES A DEGENERATE ONE. value_eq_of_agree_on_window was one of the silent-vacuity leaves: its content is that two agents may differ arbitrarily OUTSIDE the window and still agree on it, so applying it at one agent against itself yields value = value, closes by rfl, and says nothing about a window. Examples.Wireheading.AgentEquations.patient and impatient agree on every history of length at most two and disagree on the first one past it; patient_ne_impatient and utility_ne_past_window state the disagreement as theorems so it cannot quietly become vacuous, value_agrees_on_window is the factorization at that pair, and value_runHistory_agrees is the same along a one-step run of a one-bit world. SCOPE, AGAINST THE MODULES THIS JOINS RATHER THAN AGAINST PRINT. No printed source states a relation between an interaction history and an observed history, because neither paper has the other's carrier, so there is no statement to grade against and this record carries no match grade. The transport is stated at arbitrary Action, Obs and State with no finiteness, no decidability and no instances, as wide as AgentEquations' own recursion after the 2026-09-13 widenings. value_runHistory_eq_of_agree_on_window is narrower than AgentEquations.value_eq_of_agree_on_window in exactly one way and deliberately: it fixes the history to one a determined run produced, where the general theorem quantifies over every list. That is not a weakening - the general theorem stays in the tree unchanged and is still the headline - it is the instance that has content, and the general theorem is recovered by taking any history at all.", "original_source_refs": [], "related_result_ids": [ "LAND-WIRE-OBJ-001", "LAND-DEC-MDP-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AgentEquations.toDecisionHistory", "AgentEquations.ofDecisionHistory", "AgentEquations.ofDecisionHistory_toDecisionHistory", "AgentEquations.toDecisionHistory_ofDecisionHistory", "AgentEquations.toDecisionHistory_injective", "AgentEquations.runHistory", "AgentEquations.runHistory_length", "AgentEquations.value_runHistory_eq_of_agree_on_window" ], "module": "AISafetyAtlas.Wireheading.AgentHistory", "build_command": "lake build AISafetyAtlas.Wireheading.AgentHistory AISafetyAtlas.Examples.Wireheading.AgentEquations" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Wireheading.AgentEquations.value_runHistory_eq_of_agree_on_window", "type": "NEW_PROOF", "source_declarations": [ "AgentEquations.toDecisionHistory", "AgentEquations.ofDecisionHistory", "AgentEquations.ofDecisionHistory_toDecisionHistory", "AgentEquations.toDecisionHistory_ofDecisionHistory", "AgentEquations.toDecisionHistory_injective", "AgentEquations.runHistory", "AgentEquations.runHistory_length", "AgentEquations.value_runHistory_eq_of_agree_on_window" ], "application": "Interoperability: the Ring-Orseau value recursion evaluated along a run of the shared Decision carrier, so the window is counted in steps taken. The hypothesis is agreement on every history of length at most n+m, not only on the histories the run can reach." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "An agent's horizon can be measured in steps it actually took, not in lists nobody produced.", "use": "Scoring a real run by an agent's own value recursion.", "title": "Histories on the shared carrier", "attribution": "Ring & Orseau; Turner & Tadepalli" } }, { "id": "LAND-WIRE-GOALCARRIER-001", "name": "Self-modification on the Decision carrier", "tags": [ "agent-incentives" ], "notes": "A SECOND JOIN, AND THE MODEL THE FIRST ONE WAS MISSING. GoalPreservation states goal preservation over a Model History WorldAction PolicyName in which ALL THREE TYPES ARE OPAQUE PARAMETERS. That is the right generality for the theorem and the wrong thing to leave alone: an opaque carrier is one nothing can be chained to, and the only model of it in the tree collapsed every parameter to Unit, where names_surjective is free because there is one target, OptimalAt is vacuous because there is one candidate, and a constant-zero continuation makes the Bellman identity 0 = 0. Goal preservation held there and said nothing. WHAT THE MODEL IS. scheduleModel's history type is AISafetyAtlas.Decision.History, the carrier the rest of the tree runs policies along. A POLICY NAME IS A SCHEDULE, an infinite sequence of actions, and acting plays the head and becomes the tail - the smallest honest reading of self-modification here, since the agent's next self is a different named policy chosen by the current one and nothing outside supplies it. names_surjective holds by scheduleAct_surjective, consing the action onto the schedule, rather than by collapse. continuation is scheduleValue, the discounted sum of d^k u(p k), which converges by scheduleValue_summable at a discount below one and a bounded utility - both the source's own conditions, neither added to make the sum behave. coherent is scheduleValue_bellman, obtained by splitting the first term off a convergent series, so it is a theorem about a series and not a definition restated. THE WITNESS CARRIES ITS OWN NON-DEGENERACY. Examples bitModel is the model at one bit: acting is worth one, not acting is worth nothing, the discount is a half. scheduleValue_never and scheduleValue_always compute the two schedules at 0 and 2, scheduleValue_never_lt_always separates them, and not_optimal_of_never_act shows the candidate set has more than one element in it - without which bitModel_run_optimal and bitModel_goal_preservation would again be claims about a set with one member. bitModel_step_length shows the run moves along the carrier rather than standing still, and bitModel_names_surjective exhibits the surjection. WHY IT WAS BUILT. Two private elicitation notes (2026-08) name seven incompatible notions of agent behaviour in this tree with no transports between any of them and call that a present-tense defect: interoperable, not silos. CRMDP reached the shared carrier first, AgentHistory second, this third. WHAT IT DOES NOT CLAIM. Not a reward and not an MDP: the world appends the action and the observation the action produces, with no transition kernel and no state; CRMDP is where a reward lives on this carrier. THE UTILITY IGNORES THE HISTORY, deliberately, so the discounted return is a function of the schedule alone and the Bellman identity is a statement about series rather than about the world - a history-dependent utility is a different and harder model, and nothing here rules it out or supplies it. Optimality is not proved in the library module: scheduleValue_le_of_isMax is the ingredient, and inhabiting OptimalAt needs an action maximizing u, which is a hypothesis about u and lands in Examples. GoalPreservationSource, the fourth of the seven, carries a Percept field and is NOT joined here.", "original_source_refs": [], "related_result_ids": [ "LAND-GOAL-001", "LAND-WIRE-AGENTHISTORY-001", "LAND-DEC-MDP-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "GoalPreservation.stepHistory", "GoalPreservation.length_stepHistory", "GoalPreservation.scheduleAct", "GoalPreservation.scheduleAct_surjective", "GoalPreservation.scheduleValue", "GoalPreservation.scheduleValue_summable", "GoalPreservation.scheduleValue_bellman", "GoalPreservation.scheduleValue_le_of_isMax", "GoalPreservation.scheduleModel", "GoalPreservation.scheduleModel_optimalAt" ], "module": "AISafetyAtlas.Wireheading.GoalPreservationCarrier", "build_command": "lake build AISafetyAtlas.Wireheading.GoalPreservationCarrier AISafetyAtlas.Examples.Wireheading.GoalPreservationCarrier" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Wireheading.GoalPreservation.scheduleModel_optimalAt", "type": "NEW_PROOF", "source_declarations": [ "GoalPreservation.stepHistory", "GoalPreservation.length_stepHistory", "GoalPreservation.scheduleAct", "GoalPreservation.scheduleAct_surjective", "GoalPreservation.scheduleValue", "GoalPreservation.scheduleValue_summable", "GoalPreservation.scheduleValue_bellman", "GoalPreservation.scheduleValue_le_of_isMax", "GoalPreservation.scheduleModel", "GoalPreservation.scheduleModel_optimalAt" ], "application": "Interoperability: goal preservation's opaque history carrier instantiated at AISafetyAtlas.Decision.History, with a discounted return as the continuation, so the theorem is stated about a model in which policies differ." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "An agent that rewrites itself can keep the goal it started with.", "use": "Checking a self-modification argument against a model where policies actually differ.", "title": "Self-modification on the shared carrier", "attribution": "Everitt et al.; Turner & Tadepalli" } }, { "id": "LAND-PREF-TRAJECTORY-001", "name": "Preference unidentifiability at a trajectory", "tags": [ "agent-incentives" ], "notes": "THE DEGENERACY, ABOUT WHAT IS OBSERVED RATHER THAN ABOUT A FUNCTION. AISafetyAtlas.Preference states Armstrong and Mindermann's result at Policy S A = S -> A: fix an observed policy and every reward is still consistent with it. That is the source's own reading and as a statement about a function it is complete. It is also unobservable - nobody sees a policy, what is seen is a trajectory, the actions taken and the observations that followed - and the two are different objects. Until a Preference policy could be RUN there was no way to ask whether the degeneracy survives the difference. THE BRIDGE. ofMemoryless embeds a memoryless policy into AISafetyAtlas.Decision.DetPolicy by reading the last observation, and ofMemoryless_injective shows the carrier identifies nothing Preference distinguishes. detStateAt_ofMemoryless_succ is the content: the induced run obeys the memoryless recursion, so no part of the history beyond the last observation is consulted and this is an embedding of a memoryless agent rather than a re-encoding that quietly gains memory. THE STATEMENT THAT NEEDED THE BRIDGE. consistent_rewards_of_trajectory_eq_univ: the set of rewards consistent with an observed run is ALL of them. Fix a world, a start, and everything the agent was seen to do for as many steps as anyone watches; the rewards this rules out are none. THE PROOF IS SHORT AND THAT IS THE POINT - the difficulty was never the argument, it was that the statement could not be WRITTEN while a policy and a run lived in different trees. THE WITNESS FIXES A WORLD WHERE TRAJECTORIES DIFFER. Read at a world where every policy produces the same run the theorem would be true and empty. Examples flipWorld is one bit of state which the action overwrites, seen by the agent; copy repeats what it sees and contradict negates it, copy_ne_contradict and ofMemoryless_copy_ne_contradict separate them, and run_differs shows they are in different states after one step. So the run genuinely reports the policy, and the degeneracy still says the reward cannot be read off it. SCOPE, AGAINST THE MODULES THIS JOINS. No printed source states a relation between a memoryless policy and a run over an observed history, so there is nothing to grade against and this record carries no match grade. MEMORYLESS POLICIES ONLY: a history policy that genuinely consults its past is not in the image of ofMemoryless, and nothing here says the degeneracy extends to one - it does, trivially, by the same planner, and what is absent is a reason to state it, since Preference's object is memoryless. Determined runs only; the drawn run is a PMF and is untouched. NO OPTIMALITY ANYWHERE: a planner here is unconstrained exactly as in AISafetyAtlas.Preference, so nothing is maximized and nothing follows about inverse reinforcement learning under a rationality assumption. WHY IT WAS BUILT. Two private elicitation notes (2026-08) name seven incompatible notions of agent behaviour in this tree with no transports between any of them and call that a present-tense defect: interoperable, not silos. CRMDP reached the shared carrier first, AgentHistory second, GoalPreservationCarrier third, this fourth.", "original_source_refs": [], "related_result_ids": [ "BY-011", "LAND-DEC-MDP-001", "LAND-WIRE-AGENTHISTORY-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Preference.lastObs", "Preference.lastObs_append", "Preference.ofMemoryless", "Preference.ofMemoryless_injective", "Preference.lastObs_detHistoryUpTo", "Preference.detStateAt_ofMemoryless_succ", "Preference.trajectory_reward_unidentifiable", "Preference.consistent_rewards_of_trajectory_eq_univ" ], "module": "AISafetyAtlas.Preference.Trajectory", "build_command": "lake build AISafetyAtlas.Preference.Trajectory AISafetyAtlas.Examples.Preference.Trajectory" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Preference.consistent_rewards_of_trajectory_eq_univ", "type": "NEW_PROOF", "source_declarations": [ "Preference.lastObs", "Preference.lastObs_append", "Preference.ofMemoryless", "Preference.ofMemoryless_injective", "Preference.lastObs_detHistoryUpTo", "Preference.detStateAt_ofMemoryless_succ", "Preference.trajectory_reward_unidentifiable", "Preference.consistent_rewards_of_trajectory_eq_univ" ], "application": "Interoperability: preference unidentifiability stated about an observed trajectory in a world rather than about a policy function, by running a memoryless policy on the shared Decision carrier." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "When any planner is admissible, watching an agent for any length of time rules out no reward at all.", "use": "Bounding what behaviour alone can tell you about what an agent wants.", "title": "Unidentifiability at a trajectory", "attribution": "Armstrong & Mindermann" } }, { "id": "LAND-VERIF-ROBOTRUN-001", "name": "A verified behaviour, run", "tags": [ "verification" ], "notes": "A VERIFICATION VERDICT, CARRIED ONTO A TRAJECTORY. Verification.Robot states the undecidability result over Behavior Scenario Action = Scenario -> N -> Action, a total action trace indexed by scenario and operational cycle. NOTHING IN THE TREE EVER RAN ONE. A behaviour was a function from cycles to actions and a run was a different object in a different cluster, so 'an agent whose observations are corrupted AND whose policy is verified' - the sentence a private elicitation note says cannot be written today - had no type to be written at. A CYCLE IS THE LENGTH OF THE HISTORY SO FAR. AISafetyAtlas.Decision.History is an initial observation and a list, so the cycle count is already in the carrier: at a history of length n the induced policy plays what the behaviour does at cycle n. ofBehavior is that, actionAt_ofBehavior is the identification, and detStateAt_ofBehavior_succ is the step. WHAT IT BUYS. alwaysSatisfies_run carries the verifier's acceptability predicate onto the run: a program the verifier accepts produces only acceptable actions along every determined run of every world. That is the first statement in this tree connecting a verification verdict to a trajectory. A SHARED LEMMA WAS PROMOTED RATHER THAN COPIED. detHistoryUpTo_length - a determined run of n steps records exactly n of them - now lives in Decision.MDP beside the stochastic run_history_length, and Wireheading.AgentHistory.runHistory_length calls it instead of re-proving it. Two consumers count steps where the carrier counts list entries, and they count them once. THE WITNESS CARRIES ITS OWN NON-DEGENERACY. Examples mustAlternate is a requirement some behaviours fail: alternate meets it, stubborn does not (stubborn_not_alwaysSatisfies), and behaviours_differ separates the two programs. run_moves shows the world does not stand still, so the cycle count is counting something. Read at a predicate accepting everything, verified_run_is_acceptable would be true and empty. WHAT IT DOES NOT CLAIM. OPEN LOOP: a Behavior does not read observations, so the induced policy ignores everything except how long it has been running - that is the source's object, not a simplification introduced here, and a behaviour that reacts is a different type nothing here supplies. NO UNDECIDABILITY IS RE-PROVED: action_safety_unverifiable is untouched; this carries its PREDICATE to a run, not its proof. Determined runs only. THE SCENARIO IS A PARAMETER, NOT A WORLD: ofBehavior fixes a scenario and the world is given separately by a transition and an observation map, and nothing here says the two agree - relating a Scenario to a State is a modelling choice deliberately left open. WHY IT WAS BUILT. Two private elicitation notes (2026-08) name seven incompatible notions of agent behaviour with no transports between them and call that a present-tense defect. CRMDP reached the shared carrier first, AgentHistory second, GoalPreservationCarrier third, Preference.Trajectory fourth, this fifth.", "original_source_refs": [], "related_result_ids": [ "LAND-DEC-MDP-001", "LAND-PREF-TRAJECTORY-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Decision.detHistoryUpTo_length", "Verification.Robot.ofBehavior", "Verification.Robot.actionAt_ofBehavior", "Verification.Robot.detStateAt_ofBehavior_succ", "Verification.Robot.alwaysSatisfies_run" ], "module": "AISafetyAtlas.Verification.RobotRun", "build_command": "lake build AISafetyAtlas.Verification.RobotRun AISafetyAtlas.Examples.Verification.RobotRun" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Verification.Robot.alwaysSatisfies_run", "type": "NEW_PROOF", "source_declarations": [ "Decision.detHistoryUpTo_length", "Verification.Robot.ofBehavior", "Verification.Robot.actionAt_ofBehavior", "Verification.Robot.detStateAt_ofBehavior_succ", "Verification.Robot.alwaysSatisfies_run" ], "application": "Interoperability: a verifier's acceptability predicate carried from a behaviour's operational cycles onto every step of a determined run on the shared Decision carrier." } ] }, "public": { "group": "Limits of verification", "summary": "A robot that passes its check keeps passing it while it runs.", "use": "Turning a verification verdict into a claim about what an agent does.", "title": "A verified behaviour, run", "attribution": "Fisher, Dennis & Webster" } }, { "id": "LAND-COMP-RUNTRACES-001", "name": "The trace system a policy set generates", "tags": [ "compositionality" ], "notes": "WHERE A TRACE COMES FROM. Compositional.Hyperproperties states Clarkson and Schneider's theory over TraceSystem Trace = Set Trace with no account of what produces a trace, and AISafetyAtlas.Decision runs policies and produces histories. THE TWO CLUSTERS SHARED NO VOCABULARY AT ALL: a k-safety property could not be asked about an agent, because an agent had no trace system. runTraces is that system - the histories a set of policies reaches from a start, in a world - and with it IsKSafety, IsBadObservation and the self-composition reduction apply to policies. THE PREFIX RELATION IS THE CARRIER'S OWN. HistoryPrefix is same initial observation and list extension, and detHistoryUpTo_prefix_succ says a run's stages grow along it, so a trajectory's prefixes are its earlier stages and nothing had to be invented. THE STATEMENT WORTH HAVING. bad_observation_prefixes_are_run_prefixes: an observation realized by a policy set's trace system is not an abstract set of prefixes - each member extends to a stage of a NAMED policy's run. exists_bad_trajectories_of_isKSafety is the packaged form: a k-safety violation by a policy set is witnessed by at most k PARTIAL TRAJECTORIES, which is the form a counterexample to a security property is normally reported in. BOTH HYPOTHESES ARE DISCHARGED IN THE WITNESS, NOT ASSUMED. Examples nothingHasHappened is the hyperproperty 'no trace records a step', and nothingHasHappened_isKSafety PROVES it is 1-safety rather than positing it: a violating system holds a trace of positive length, that single trace is the bad observation, and anything realizing it holds a trace at least as long. system_violates is the violation. Without both, exists_bad_trajectories_of_isKSafety would be a hypothesis chain nothing satisfies. WHAT IT DOES NOT CLAIM. Determined runs only; a trace system of distributions is a different object. COMPLETE TRACES ARE FINITE HISTORIES: Clarkson and Schneider's traces are infinite and the histories here are finite, so runTraces is the system of STAGES rather than of limits, no limit is taken, and a hyperproperty about infinite behaviour is not reached. No hyperproperty is exhibited in the library module - IsKSafety is a hypothesis throughout, and the module supplies the system, not a property of one. THE PRODUCER REACHES CRMDP AND NOT GoalPreservation: CRMDP's history type is Decision.History at its own observation alphabet so its determined runs are traces here, while GoalPreservation.Model.run steps by its own next field rather than by Decision.detRun, so its trajectories are not runTraces traces even where its histories are Decision.History. The stale non-claim in Compositional.Hyperproperties.Knowability that named three modules with no trace consumer was corrected in the same commit. WHY IT WAS BUILT. Two private elicitation notes (2026-08) name seven incompatible notions of agent behaviour with no transports between them and call that a present-tense defect: interoperable, not silos. This is the sixth to reach the shared carrier, after CRMDP, AgentHistory, GoalPreservationCarrier, Preference.Trajectory and Verification.RobotRun.", "original_source_refs": [], "related_result_ids": [ "LAND-DEC-MDP-001", "LAND-VERIF-ROBOTRUN-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Hyperproperties.HistoryPrefix", "Hyperproperties.detHistoryUpTo_prefix_succ", "Hyperproperties.runTraces", "Hyperproperties.mem_runTraces", "Hyperproperties.runTraces_nonempty", "Hyperproperties.runTraces_mono", "Hyperproperties.bad_observation_prefixes_are_run_prefixes", "Hyperproperties.exists_bad_trajectories_of_isKSafety" ], "module": "AISafetyAtlas.Compositional.TraceSystem", "build_command": "lake build AISafetyAtlas.Compositional.TraceSystem AISafetyAtlas.Examples.Compositional.TraceSystem" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Compositional.Hyperproperties.exists_bad_trajectories_of_isKSafety", "type": "NEW_PROOF", "source_declarations": [ "Hyperproperties.HistoryPrefix", "Hyperproperties.detHistoryUpTo_prefix_succ", "Hyperproperties.runTraces", "Hyperproperties.mem_runTraces", "Hyperproperties.runTraces_nonempty", "Hyperproperties.runTraces_mono", "Hyperproperties.bad_observation_prefixes_are_run_prefixes", "Hyperproperties.exists_bad_trajectories_of_isKSafety" ], "application": "Interoperability: hyperproperty theory applied to agents, with a k-safety violation by a policy set reported as at most k partial trajectories of named policies." } ] }, "public": { "group": "Composition and interpretability", "summary": "When a group of agents breaks a security property, a few partial runs show it.", "use": "Reporting a hyperproperty violation as trajectories rather than as abstract prefixes.", "title": "Traces of a policy set", "attribution": "Clarkson & Schneider" } }, { "id": "LAND-GOAL-001", "name": "Finite-percept on-policy goal-preservation induction step", "tags": [ "agent-incentives" ], "notes": "Everitt-Filan-Daswani-Hutter, AGI 2016, LNCS 9782, Theorem 12, at two abstraction levels. VERSION: the published chapter is canonical for statements and proves nothing, delegating every proof to the technical report arXiv:1605.03142, whose Theorem 16 is word-for-word the published Theorem 12 and is what older atlas notes cite. The optimal-policy-existence result those notes call Theorem 20 lives in the report's Appendix A and is invoked as the first step of the report's own proof, so the gap recorded here is real against the only proof that exists. GoalPreservationSource is the headline finite-percept one-step argument: it drops the atlas's earlier naming-surjectivity premise, compares the selected continuation only with the named initial policy, and uses a normalized full-support percept distribution to turn a strict pointwise loss into a strict expected loss. It is not the full or source-strength Theorem 16. initial_dominates and contValue are supplied rather than derived from modification independence and the source's Theorem 20; policy names and policies are not separated by an explicit naming map; expectations are finite discrete sums rather than integrals; and no all-times trajectory theorem is proved. SixTargets.finitePerceptGoalModel witnesses the interface with two percepts, a uniform full-support distribution, and a genuinely dominated continuation. The older deterministic GoalPreservation model is retained as a stronger-assumption specialization: next_policy_optimal is its load-bearing statement, run_optimal lifts it along the reached path, and goal_preservation is a corollary. Its names_surjective premise forces an infinite policy-name type with at least two world actions; SixTargets.twoActionGoalModel witnesses satisfiability. RELATED only; no off-policy preservation or AI-system bridge is claimed. Supporting self-modification theorem; no direct Table-1 coverage.", "original_source_refs": [ "atlas-ref-everitt-2016-self-modification" ], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "relationship": "RELATED", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "GoalPreservationSource.Model.selected_matches_initial", "GoalPreservationSource.Model.safe_modification", "GoalPreservationSource.Model.qValue_selected_eq_initial", "next_policy_optimal", "run_optimal", "goal_preservation" ], "module": "AISafetyAtlas.Wireheading.GoalPreservationSource", "build_command": "lake build AISafetyAtlas.Wireheading.GoalPreservation AISafetyAtlas.Wireheading.GoalPreservationSource", "scope_delta": { "summary": "A finite-percept one-step induction step, not source-strength published Theorem 12 (technical-report Theorem 16). `initial_dominates` and the continuation value are assumed rather than derived from modification independence and the technical report's Appendix-A Theorem 20, which the source's own proof invokes first and the published chapter prints no proof of; policy names and policies are not separated by an explicit naming map; and this row's declarations prove only the one-step induction step. Corrected 2026-09-10: this summary also said 'expectations are finite discrete sums rather than integrals', and the row's own name calls the step finite-percept, both of which read as narrowings. They are not. Page 3 of arXiv:1605.03142v1 fixes a finite action set and a finite percept set and restricts the environment to full-support percept distributions, so a normalized Finset sum over a Fintype of percepts is print's expectation exactly and the positivity field is print's own condition. The RELATED relationship stands on the remaining clauses, which are untouched. The all-times trajectory theorem, equation (13) itself, is LAND-GOAL-002, which iterates this step after adding print's equation (7) and the printed optimality of the initial policy - and which derives `initial_dominates` rather than assuming it.", "evidence": "docs/provenance/a1-a3-b1-b3-b7-reverification.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Wireheading.GoalPreservationSource.Model.selected_matches_initial", "type": "NEW_PROOF", "source_declarations": [ "GoalPreservationSource.Model.selected_matches_initial", "GoalPreservationSource.Model.safe_modification", "GoalPreservationSource.Model.qValue_selected_eq_initial", "next_policy_optimal", "run_optimal", "goal_preservation" ], "application": "Goal preservation: a rational agent's one-step self-modification selects a continuation matching its initial goal. Finite-percept induction step of Theorem 12 of Everitt, Filan, Daswani and Hutter, AGI 2016, LNCS 9782 (Theorem 16 of the technical report arXiv:1605.03142, which is where its proof lives)." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "An agent that can rewrite itself can still keep its original goal.", "use": "Self-modification arguments.", "title": "Goal preservation", "attribution": "Everitt, Filan, Daswani & Hutter" } }, { "id": "LAND-VRL-001", "name": "Value reinforcement learning and the consistency-preserving constraint", "tags": [ "agent-incentives" ], "notes": "Everitt and Hutter, Avoiding Wireheading with Value Reinforcement Learning, arXiv:1605.03143v1, sections 3 to 5, transcribed at print's own quantifiers. Definition 3 is condReward, marginalReward and posterior; Definition 5 is IsCP; Definitions 7, 8 and 9 are rlValue, utilityValue and vrlValue; Definitions 10 and 11 are IsUVRLAction and IsCPVRLAction; Definition 12 is IsEEP. Lemma 13 is isEEP_of_isCP and Theorem 14 is vrlValue_of_isCP: for a consistency-preserving action the VRL value function reduces to a prior-weighted mixture of utility-agent values, so the reward evidence has dropped out and no choice among such actions can be motivated by what the reward channel would report. posterior_expectation is the computation Appendix C's Lemma 27 rests on, and vrlValue_eq_rlValue is that lemma under the support condition print's equation (1) leaves implicit - unconstrained VRL is RL, which is the contrast Theorem 14 is stated against. NOT NEW MATHEMATICS: every proof is the source's. What is added is that the division in equation (1) is handled rather than assumed away. Lean's division by zero is zero, so a state-reward pair with vanishing C(r|s) but non-vanishing C(u)C(r|s,u) would falsify Lemma 13 in Lean while being unreachable in print's reading; prior_nonneg is a structure field and prior_mul_condReward_eq_zero_of_marginal_eq_zero squeezes every numerator in that case, which is why Lemma 13 needs no side condition on its statement. SCOPE: narrower than print in two named ways. Utility functions are indexed rather than extensional, so a cardinality claim about U would not transfer and none is made; the reward carrier needs decidable equality to write the indicator. WHAT PRINT ACTUALLY ASSUMES, CORRECTED 2026-09-13: page 4 reads 'We also assume that R, S, and U are finite OR COUNTABLE. Finally, to ensure well-defined expectations, we assume that R is bounded if it is countable.' An earlier revision of this note said print says only 'a class U of utility functions S to R' and that print's S and R are finite already; both came from a text extraction that wrapped 'or countable' onto a line the search did not reach, and both are wrong. FINITENESS AXIS CLOSED 2026-09-13, by exactly that route. Every sum in the module - the reward marginal, Definition 9's value function, Definition 12, the RL and utility values, and Lemma 27's posterior expectation - is now an unconditional sum over an arbitrary type, and the variable block carries no Fintype at all. NOTHING WAS ADDED TO Beliefs: the structure is unchanged, so no axis was traded for this one. The two facts print's setup supplies are hypotheses of the results that need them and of no others - Summable of the utility prior, for the squeeze lemma and Lemma 13, and summability of the joint family Theorem 14 exchanges. At a Fintype each is Summable.of_finite, so the signatures carried before today are instances of the new ones and Examples.Wireheading.ValueLearning discharges them that way. Five audit cells in section 12 went Narrower to Same: Definition 3 with equation (1), Definition 9, Definition 12, Lemma 13 and Theorem 14. The Lemma 27 row stays Narrower on the support condition, which is a different axis and is print's own implicit division hypothesis. WHAT THIS WIDENING DOES NOT PROVE: the unconditional sum is zero by convention at a non-summable family, so where print's boundedness clause fails this module states something that is not print's equation; and the exchange hypothesis is stated as summability of the joint family rather than derived from print's countability and bounded R, which is a derivation this tree does not carry. NON-CLAIMS: one decision step only, no sequential agent, no Definition 2, no Assumption 4, no Assumption 15, and none of the appendix material - Definitions 17, 21 and 26, Lemma 20, Theorems 22 and 25, Corollary 24. Assumption 6 (a CP action exists) is print's assumption and is never taken as a standing axiom; AISafetyAtlas.Examples.Wireheading.ValueLearning inhabits it and, in the same two-state model, exhibits a non-CP action with strictly higher VRL value, so the constraint is what removes the incentive rather than a restriction on an empty set. Nothing here claims a useful prior consistent with a real reward channel can be specified, which the source's own Table 1 caption records as open. Supporting wireheading mathematics, not itself a Table-1 row.", "original_source_refs": [ "atlas-ref-everitt-hutter-2016-vrl" ], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "relationship": "RELATED", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "ValueLearning.Beliefs.isEEP_of_isCP", "ValueLearning.Beliefs.vrlValue_of_isEEP", "ValueLearning.Beliefs.vrlValue_of_isCP", "ValueLearning.Beliefs.vrlValue_eq_prior_mixture", "ValueLearning.Beliefs.posterior_expectation", "ValueLearning.Beliefs.vrlValue_eq_rlValue", "ValueLearning.Beliefs.sum_condReward", "ValueLearning.Beliefs.marginalReward_nonneg" ], "module": "AISafetyAtlas.Wireheading.ValueLearning", "build_command": "lake build AISafetyAtlas.Wireheading.ValueLearning AISafetyAtlas.Examples.Wireheading.ValueLearning", "scope_delta": { "summary": "The single-step decision problem of sections 3 to 5 only, over a finite utility class. Print's `U` is not assumed finite; here every sum over utility functions is a `Finset` sum, which narrows Definitions 3, 9 and 12 and Lemma 13 and Theorem 14 with them. Utility functions are indexed rather than extensional. Nothing iterates, and no appendix result is reproduced.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Wireheading.ValueLearning.Beliefs.vrlValue_of_isCP", "type": "NEW_PROOF", "source_declarations": [ "ValueLearning.Beliefs.isEEP_of_isCP", "ValueLearning.Beliefs.vrlValue_of_isEEP", "ValueLearning.Beliefs.vrlValue_of_isCP", "ValueLearning.Beliefs.vrlValue_eq_prior_mixture", "ValueLearning.Beliefs.posterior_expectation", "ValueLearning.Beliefs.vrlValue_eq_rlValue", "ValueLearning.Beliefs.sum_condReward", "ValueLearning.Beliefs.marginalReward_nonneg" ], "application": "Wireheading: constraining an agent to actions on which its reward-channel belief agrees with the reward marginal its own utility prior induces removes the reward evidence from its value function, so it has no incentive to corrupt the channel. Theorem 14 of Everitt and Hutter, arXiv:1605.03143v1." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "An agent told to keep its beliefs consistent stops wanting to fool its own sensor.", "use": "Designing a reward channel an agent has no reason to corrupt.", "title": "Value reinforcement learning", "attribution": "Everitt & Hutter" } }, { "id": "LAND-GOAL-002", "name": "Theorem 16 equation (13) over the whole trajectory, without naming surjectivity", "tags": [ "agent-incentives" ], "notes": "Everitt, Filan, Daswani and Hutter, arXiv:1605.03142v1, Theorem 16 (published AGI 2016 LNCS 9782 Theorem 12), conclusion equation (13): for every t, for all percept sequences, and for the on-policy action sequence, the Q-value of the action the current self-modified policy takes equals the Q-value of the action the initial policy would have taken. equation_thirteen is that conclusion. WHAT THIS ADDS TO LAND-GOAL-001: the atlas had the two halves and not the whole. Wireheading.GoalPreservation reaches equation (13) but only under names_surjective - every world-action and next-policy pair is emitted by some NAMED policy - which is not print's premise: print maximises over the full policy space, a function space into the product of world actions and names, so every pair is realised for free, while P is only the set of names and the source says explicitly that not every policy has one. Requiring the named policies to be surjective conflates the two. Wireheading.GoalPreservationSource removes that premise from the one-step argument but never iterates it, because nothing in its Model ties contValue to qValue. GoalPreservationRun.Model supplies the missing link as two printed fields - bellman, which is print's equation (7), and initial_optimal, which is print's hypothesis that the initial policy optimises the realistic value function - and iterates the one-step lemma to the printed conclusion. GoalPreservationSource's initial_dominates stops being an assumption: contValue_le_initial derives it. STILL OWED ON THIS SOURCE, unchanged from LAND-GOAL-001: modification independence is built into the signature rather than proved, and the technical report's Appendix-A Theorem 20 (optimal policy existence) and Theorem 21 (optimal policy name), which its own proof of Theorem 16 invokes first, are not reproduced - initial_optimal asserts what they would supply. Expectations are normalized Finset sums over a Fintype of percepts, not integrals. Corrected 2026-09-10: this used to end 'so this is the finite-percept case', which implied print has a wider one. It does not - page 3 of arXiv:1605.03142v1 fixes a finite action set and a finite percept set and restricts the environment to full-support percept distributions, so the Fintype and the positivity field are print's own conditions and neither is a narrowing. Policy modification only, never utility modification; print's Theorems 14 and 15 are formalized nowhere in the atlas. AISafetyAtlas.Examples.Wireheading.GoalPreservationRun inhabits both new fields with a model that genuinely alternates policy names, so equation (13) there compares the values of two DIFFERENT action pairs rather than being a reflexivity, and in which contValue separates a dominated name, so contValue_le_initial is strict somewhere. MISSING-OBJECT AXIS CLOSED 2026-09-13: Definition 3 on page 5 of arXiv:1605.03142v1 (sha256 1b7a3c09..., read as a rendered image) reads 'Let Pi = {(A x E)* -> A} be the set of all policies, and let iota : P -> Pi assign names to policies', so print's Pi is the full function space from histories to actions and its iota is exactly the atlas's act uncurried. Both models now name them - Policy and Model.name, with name_eq_act recording that naming them changed nothing - so print's quadruple has all four components present as objects. The example exhibits a model whose naming map is NOT surjective, which is print's own 'some policies will necessarily lack names', so the absence of a surjectivity assumption is not vacuous. The audit's Definition 3 row went Narrower to Same and stays Partial on a coverage axis: print's argument that P = Pi is impossible is a cardinality claim, and what is carried is one model where iota misses a policy, not that every model must. Supporting self-modification mathematics, not itself a Table-1 row.", "original_source_refs": [ "atlas-ref-everitt-2016-self-modification" ], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "relationship": "RELATED", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "GoalPreservationRun.Model.equation_thirteen", "GoalPreservationRun.Model.run_optimal", "GoalPreservationRun.Model.optimalAt_next", "GoalPreservationRun.Model.contValue_le_initial", "GoalPreservationRun.Model.run_contValue_eq_initial" ], "module": "AISafetyAtlas.Wireheading.GoalPreservationRun", "build_command": "lake build AISafetyAtlas.Wireheading.GoalPreservationRun AISafetyAtlas.Examples.Wireheading.GoalPreservationRun", "scope_delta": { "summary": "The printed conclusion at the printed quantifiers, minus the two things print proves in its own appendix: optimal-policy existence and optimal-policy nameability, which `initial_optimal` asserts instead. Modification independence is a consequence of the signature rather than a derivation. Percepts are finite and expectations are normalized `Finset` sums - corrected 2026-09-10, this clause used to sit here as a narrowing and is not one: print fixes a finite percept set on page 3 and requires full-support percept distributions, so both are print's own conditions. docs/provenance/source-coverage-audit.md regraded eq. (13) and eq. (14) from Narrower to Same on that retraction. This row stays RELATED for the two appendix results it assumes rather than proves, which is a different axis and is unchanged.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Wireheading.GoalPreservationRun.Model.equation_thirteen", "type": "NEW_PROOF", "source_declarations": [ "GoalPreservationRun.Model.equation_thirteen", "GoalPreservationRun.Model.run_optimal", "GoalPreservationRun.Model.optimalAt_next", "GoalPreservationRun.Model.contValue_le_initial", "GoalPreservationRun.Model.run_contValue_eq_initial" ], "application": "Self-modification: an agent optimising a realistic value function self-modifies only into policies that are worth exactly what its initial policy was worth, at every step of every percept sequence. Equation (13) of Theorem 16 of Everitt, Filan, Daswani and Hutter, arXiv:1605.03142v1." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "A self-rewriting agent that plans realistically never trades its goal away, at any step.", "use": "Self-modification arguments that must run for more than one step.", "title": "Goal preservation along the trajectory", "attribution": "Everitt, Filan, Daswani & Hutter" } }, { "id": "LAND-JOINTOBS-001", "name": "Coalition-indexed joint observation and the emitted-interface coverage boundary", "tags": [ "oversight", "multi-agent" ], "result_shape": "CHARACTERIZATION", "notes": "When a hazard can be decided from what a coalition of principals is permitted to observe. EvidenceArchitecture separates private evidence from declared emitted views, and CoalitionInput enforces coalition access in the type of a candidate computation. Since the LAND-KNOW-001 kernel landed, Covers is definitionally Knowable q.observe and Refines is Determines, so the coverage laws are the generic kernel applied to q.observe rather than separate arguments, with every statement and axiom profile unchanged. What remains coalition-specific is the evidence architecture, the typed coalition restriction, the collision certificate, the certified finite checker decideCoverage over an explicit execution enumeration, and the bounded portfolio target with its inclusion-minimal and cost-optimal notions. Two worked architectures exhibit the failures and one narrow joint success. This is checked target semantics, not a synthesizer: nothing generates candidates, searches a mechanism library, or validates the cost model. Residual adds the distribution-free quantitative face of coverage: worstResidual counts how many target values the worst single output still leaves open, covers_iff_worstResidual_le_one identifies coverage with the residual-at-most-one case, and worstResidual_le_postprocess is the repair boundary as a comparable number. Residual counts executions and never weighs them, so there is no probability, prior, expectation or entropy anywhere in it. A residual of one fixes every state-attribution question; it says nothing about whether a principal reported honestly. Everything holds under the fixed truthful mechanism M0; no strategic, permissibility, accountability or normative safety claim follows.", "original_source_refs": [], "related_result_ids": [ "LAND-KNOW-001" ], "relations": [ { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "Covers and Refines are definitionally the kernel's Knowable and Determines, so Coverage and RepairBoundary discharge their proofs by calling the kernel rather than repeating a factorization argument." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "covers_iff_no_collision", "not_covers_of_collisionWitness", "exists_collisionWitness_of_not_covers", "decideCoverage", "decideCoverage_covered_iff", "CoverageResult.covered_eq_true_iff", "decidableCovers", "postprocess_cannot_repair_collision", "covers_of_refines", "observe_truthful", "PortfolioCovers", "PortfolioIndistinguishable", "portfolioCovers_implies_hazardEquivalent", "InclusionMinimalCovering", "PortfolioCost", "CostOptimalCovering", "inclusionMinimal_of_costOptimal", "residual", "worstResidual", "covers_iff_worstResidual_le_one", "covers_iff_residual_le_one", "worstResidual_le_postprocess" ], "module": "AISafetyAtlas.Oversight.JointObservation", "build_command": "lake build AISafetyAtlas.Oversight.JointObservation AISafetyAtlas.Examples.Oversight.JointObservation.Procurement AISafetyAtlas.Examples.Oversight.JointObservation.Portfolio" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Oversight.JointObservation.covers_iff_no_collision", "type": "NEW_PROOF", "source_declarations": [ "covers_iff_no_collision", "not_covers_of_collisionWitness", "exists_collisionWitness_of_not_covers", "decideCoverage", "decideCoverage_covered_iff", "CoverageResult.covered_eq_true_iff", "decidableCovers", "postprocess_cannot_repair_collision", "covers_of_refines", "observe_truthful", "PortfolioCovers", "PortfolioIndistinguishable", "portfolioCovers_implies_hazardEquivalent", "InclusionMinimalCovering", "PortfolioCost", "CostOptimalCovering", "inclusionMinimal_of_costOptimal" ], "application": "Oversight: a hazard is decidable from what a coalition of principals may observe exactly when no two executions collide on that coalition's view. In-tree model of emitted interfaces and coalition access." } ] }, "public": { "group": "What an observer can recover", "summary": "A group sees only what its members share. Processing cannot recover the rest.", "use": "Designing oversight that can actually work.", "title": "Coalition coverage", "attribution": "Workbench infrastructure" } }, { "id": "LAND-SELFREF-001", "name": "The self-model as a component of the state it models", "tags": [ "information-theory", "interpretability" ], "result_shape": "CHARACTERIZATION", "notes": "The layer where the observation stops being an arbitrary map. Every earlier knowability module states in its non-claims that nothing makes the observation a projection of a state containing the observer; here the observer's model IS a component of the global state, written Model x Rest, and selfRead is the projection onto it. Self-reference is then forced rather than assumed: the target is the whole state, which includes the Model component, so a complete self-model must model itself. selfComplete_iff_subsingleton_rest says complete self-knowledge holds if and only if the remainder is a subsingleton — achievable, but only degenerately, which is what makes it a characterization rather than a blanket denial. not_selfComplete_of_two_rest is the axiom-free direction, exhibiting the colliding pair. card_rest_le_one_of_selfComplete is the counting form and is proved through LAND-AMBIG-001's card_image_le_of_knowable rather than by a fresh argument: the state carries |Model| * |Rest| values while the reading carries at most |Model|, so completeness is a cardinality inequality only |Rest| <= 1 can satisfy. That makes it the first consumer of the ambiguity layer outside Examples. AISafetyAtlas.Examples.Knowledge.SelfReference inhabits both directions, including the degenerate one, so the iff is not an impossibility in disguise. Explicit non-claims: this is NOT Breuer's theorem, which is LAND-SELFMEAS-002 and models the apparatus with an inference map and meshing rather than assuming a product decomposition; there is no regress or hierarchy over models of models, the obstruction being one level deep; there is no resource bound, budget or process count; and there is no dynamics, so nothing here says the state changes while being modelled — composing with LAND-TEMPORAL-001 is the obvious next step and is deliberately not taken. Model and Rest are arbitrary types and are not interpreted as memory, computation or belief. Nothing here concerns consciousness: incompleteness of a self-model is a statement about a projection, and every partially observed embedded system has it, which is exactly why it cannot be evidence of anything phenomenal. No AI-system reading follows without a separate reviewed bridge.", "original_source_refs": [], "related_result_ids": [ "LAND-KNOW-001", "LAND-AMBIG-001", "LAND-SELFMEAS-002" ], "relations": [ { "target": "LAND-KNOW-001", "kind": "INSTANTIATES", "note": "Takes the state to be Model times Rest and the observation to be the first projection, so the observer is a component of what it observes." }, { "target": "LAND-AMBIG-001", "kind": "BUILDS_ON", "note": "card_rest_le_one_of_selfComplete is proved through card_image_le_of_knowable rather than by a separate cardinality argument." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.SelfReference.SelfState", "AISafetyAtlas.Knowledge.SelfReference.selfRead", "AISafetyAtlas.Knowledge.SelfReference.SelfComplete", "AISafetyAtlas.Knowledge.SelfReference.selfComplete_iff_subsingleton_rest", "AISafetyAtlas.Knowledge.SelfReference.not_selfComplete_of_two_rest", "AISafetyAtlas.Knowledge.SelfReference.card_rest_le_one_of_selfComplete" ], "module": "AISafetyAtlas.Knowledge.SelfReference", "atlas_module": "AISafetyAtlas.Knowledge.SelfReference", "build_command": "lake build AISafetyAtlas.Knowledge.SelfReference" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.SelfReference.selfComplete_iff_subsingleton_rest", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.SelfReference.not_selfComplete_of_two_rest", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.SelfReference.card_rest_le_one_of_selfComplete", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "What an observer can recover", "summary": "A system can fully know itself only if there is nothing else to know.", "use": "The precise sense in which nothing can fully model itself.", "title": "Self-modelling", "attribution": "Workbench infrastructure" } }, { "id": "LAND-ACCUM-001", "name": "Window ambiguity: accumulation bounds over a set of targets", "tags": [ "information-theory", "oversight" ], "result_shape": "BOUND", "notes": "Extends LAND-AMBIG-001 from one target to a window of them. Widening a window never reduces ambiguity and never exceeds the product of its steps, which brackets accumulation between never decreases and at most multiplies; one unknowable step makes the whole window unknowable. A second axis runs the other way: under Temporal.EvidenceMonotone a later reading determines an earlier one, so reading late lowers ambiguity while widening raises it, and the two compose. That is what makes the dependency on LAND-TEMPORAL-001 real rather than conceptual. Growth itself is deliberately not proved — it depends on whether each step adds a distinction the observation cannot see, which needs dynamics this layer does not have. AISafetyAtlas.Examples.Knowledge.Accumulation exhibits both extremes, a blind observer meeting the product ceiling and an informed one that never accumulates, so neither is mistaken for a law. Scope: the target is fixed while evidence accumulates; a target that moves with time is not covered. Finite counting only, with no probability, entropy or rate.", "original_source_refs": [], "related_result_ids": [ "LAND-AMBIG-001", "LAND-TEMPORAL-001", "LAND-KNOW-001" ], "relations": [ { "target": "LAND-AMBIG-001", "kind": "BUILDS_ON", "note": "The window bounds are stated and proved in terms of ambiguity and its fibre image." }, { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "not_knowable_pairTarget_of_not_knowable is Knowable.mono against the projection." }, { "target": "LAND-TEMPORAL-001", "kind": "BUILDS_ON", "note": "ambiguity_le_of_evidenceMonotone consumes Temporal.EvidenceMonotone: cumulative evidence makes the earlier observation a post-processing of the later one, so ambiguity_le_of_comp applies along time." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.pairTarget", "AISafetyAtlas.Knowledge.ambiguity_le_pairTarget_left", "AISafetyAtlas.Knowledge.ambiguity_le_pairTarget_right", "AISafetyAtlas.Knowledge.ambiguity_pairTarget_le_mul", "AISafetyAtlas.Knowledge.not_knowable_pairTarget_of_not_knowable", "AISafetyAtlas.Knowledge.ambiguity_le_of_evidenceMonotone", "AISafetyAtlas.Knowledge.ambiguity_le_pairTarget_of_evidenceMonotone" ], "module": "AISafetyAtlas.Knowledge.Accumulation", "atlas_module": "AISafetyAtlas.Knowledge.Accumulation", "build_command": "lake build AISafetyAtlas.Knowledge.Accumulation" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.ambiguity_le_pairTarget_left", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.ambiguity_le_pairTarget_right", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.ambiguity_pairTarget_le_mul", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.not_knowable_pairTarget_of_not_knowable", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.ambiguity_le_of_evidenceMonotone", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.ambiguity_le_pairTarget_of_evidenceMonotone", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "What an observer can recover", "summary": "Asking about more of a system can only leave more possibilities open.", "use": "Bounding how far a monitor's picture can drift.", "title": "Ambiguity over a window", "attribution": "Workbench infrastructure" } }, { "id": "LAND-AMBIG-001", "name": "Finite fibre ambiguity and the counting obstruction", "tags": [ "information-theory", "oversight" ], "result_shape": "CHARACTERIZATION", "notes": "Quantitative refinement of LAND-KNOW-001 on finite state spaces. ambiguity r f i counts the distinct target values still possible once the observation reads i; 1 is exact knowledge, 0 marks an observation value nothing realizes, and larger is the shortfall. The exactness condition is at most one rather than exactly one, because requiring equality would make every model with an unreachable observation value unknowable. knowable_iff_ambiguity_le_one is recorded and labelled a BRIDGE, not a result: it is knowable_iff_no_collision in counting form, and a module stopping there would be a cardinality wrapper around a theorem the kernel already proves. The content is the two bounds. card_image_le_of_knowable says a knowable target needs an observation with at least as many distinct outcomes as the target has values — a counting obstruction that never names a colliding pair, with not_knowable_of_card_lt as the usable contrapositive. ambiguity_le_of_comp says post-processing never decreases ambiguity anywhere, which upgrades not_knowable_comp from 'the yes/no answer cannot be repaired' to 'the shortfall never shrinks'. AISafetyAtlas.Examples.Knowledge.Ambiguity keeps the bounds honest with three models: the counting test firing where no pair is exhibited; a model where both cardinalities are 2 and the target is still unknowable, so the counting test is sufficient and never necessary; and a strict increase of ambiguity under coarsening, so the monotonicity bound is not an equality in disguise. Every declaration here is classical, inherited from Mathlib's Finset and Fintype development rather than from the statements, which differs from the axiom-free kernel core and is recorded in the module. Explicit non-claims: this counts possibilities and does not weigh them. There is no probability, entropy, rate or measure. A conditional-entropy treatment would need Shannon entropy over general probability spaces, which the pinned Mathlib does not carry — it has topological entropy, binary entropy and Kullback-Leibler divergence, and no conditional Shannon entropy — so that is a separate foundation, not an extension of this module. worstAmbiguity condenses the per-value counts into one scalar per observation, which is what a consumer comparing two observations needs, and worstAmbiguity_le_of_comp carries the coarsening bound to that scalar. No AI-system reading follows without a separate reviewed bridge.", "original_source_refs": [], "related_result_ids": [ "LAND-KNOW-001" ], "relations": [ { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "knowable_iff_ambiguity_le_one is proved through knowable_iff_no_collision." }, { "target": "LAND-KNOW-001", "kind": "REFINES", "note": "Replaces the qualitative iff with a count of the answers a single observation leaves open, at the cost of finiteness and decidable equality. Knowability becomes ambiguity at most one." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.fibre", "AISafetyAtlas.Knowledge.ambiguity", "AISafetyAtlas.Knowledge.knowable_iff_ambiguity_le_one", "AISafetyAtlas.Knowledge.not_knowable_of_one_lt_ambiguity", "AISafetyAtlas.Knowledge.card_image_le_of_knowable", "AISafetyAtlas.Knowledge.not_knowable_of_card_lt", "AISafetyAtlas.Knowledge.ambiguity_le_of_comp", "AISafetyAtlas.Knowledge.ambiguity_eq_zero_of_not_mem_image", "AISafetyAtlas.Knowledge.worstAmbiguity", "AISafetyAtlas.Knowledge.ambiguity_le_worstAmbiguity", "AISafetyAtlas.Knowledge.knowable_iff_worstAmbiguity_le_one", "AISafetyAtlas.Knowledge.worstAmbiguity_le_of_comp" ], "module": "AISafetyAtlas.Knowledge.Ambiguity", "atlas_module": "AISafetyAtlas.Knowledge.Ambiguity", "build_command": "lake build AISafetyAtlas.Knowledge.Ambiguity" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.card_image_le_of_knowable", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.not_knowable_of_card_lt", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.ambiguity_le_of_comp", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.knowable_iff_ambiguity_le_one", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "What an observer can recover", "summary": "Counts the answers an observation leaves open.", "use": "Ruling out an observability claim by counting.", "title": "Ambiguity counting", "attribution": "Workbench infrastructure" } }, { "id": "LAND-TEMPORAL-001", "name": "Time-indexed knowability, contemporaneous collisions, and delayed knowledge", "tags": [ "information-theory", "oversight" ], "result_shape": "INFRASTRUCTURE", "notes": "Adds a time index to the LAND-KNOW-001 kernel, separating the time evidence is read from the time the target refers to. KnowableFrom observe target t s is the target as of s decoded from evidence at t; KnowableAt is the contemporaneous case. That separation is the content of the module — collapsed, two different sentences become one. CollisionAt refutes contemporaneous knowledge and is axiom-free; DelayedKnowable pairs contemporaneous failure with later success, and AISafetyAtlas.Examples.Knowledge.Temporal exhibits a model inhabiting it, which is why a self-knowledge impossibility must be stated contemporaneously rather than absolutely. No result re-proves a factorization argument: knowableFrom_mono is Knowable.mono and not_knowableAt_of_collisionAt is not_knowable_of_collision. Prior art is recorded as NC-007 — the indexing is a measure-free shadow of Mathlib's filtration theory and no novelty is claimed for it; what the module has instead is that it needs no measurable structure and states cumulativity as Determines, since evidence types differ across time. Explicit non-claims are in the module docstring.", "original_source_refs": [], "related_result_ids": [ "LAND-KNOW-001", "LAND-CL-001" ], "relations": [ { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "knowableFrom_mono is Knowable.mono and not_knowableAt_of_collisionAt is not_knowable_of_collision; no factorization argument is re-proved." }, { "target": "LAND-CL-001", "kind": "BOUNDARY_PARTNER", "note": "Model delta: this row is an arbitrary indexed family of observations over an arbitrary preorder, with no dynamics and no communication; Chandy-Lamport is a concrete message-passing system with an algorithm. The pair brackets contemporaneity, not one model's frontier: DelayedKnowable says the current target can be unreadable while a later reading settles it, and the snapshot algorithm is the motivating instance of that pattern rather than a construction inside this model: nothing here is applied to it, and no theorem connects the two." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.Temporal.KnowableFrom", "AISafetyAtlas.Knowledge.Temporal.KnowableAt", "AISafetyAtlas.Knowledge.Temporal.EvidenceMonotone", "AISafetyAtlas.Knowledge.Temporal.CollisionAt", "AISafetyAtlas.Knowledge.Temporal.DelayedKnowable", "AISafetyAtlas.Knowledge.Temporal.knowableFrom_mono", "AISafetyAtlas.Knowledge.Temporal.not_knowableAt_of_collisionAt", "AISafetyAtlas.Knowledge.Temporal.collisionAt_of_not_knowableAt" ], "module": "AISafetyAtlas.Knowledge.Temporal", "atlas_module": "AISafetyAtlas.Knowledge.Temporal", "build_command": "lake build AISafetyAtlas.Knowledge.Temporal" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.Temporal.knowableFrom_mono", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Temporal.not_knowableAt_of_collisionAt", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Temporal.collisionAt_of_not_knowableAt", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "What an observer can recover", "summary": "You can learn what the state was. Not always what it is now.", "use": "Stating self-knowledge limits correctly.", "title": "Time-indexed knowability", "attribution": "shadows filtration theory; no novelty claimed for time indexing" } }, { "id": "LAND-CL-001", "name": "Chandy-Lamport distributed snapshot — termination, correctness, stable property detection", "tags": [ "multi-agent", "oversight" ], "result_shape": "ACHIEVABILITY", "notes": "Reproduced 2026-08-11 from AFP entry Chandy_Lamport (Ben Fiedler and Dmitriy Traytel, July 2020) at the pinned full AFP release 2026-02-06, Isabelle2025-2, BSD License. Session build exit 0, all seven theories, no errors; Finished Chandy_Lamport in 1:54 elapsed. The single-entry archive does not close — the Chandy_Lamport session parent is Ordered_Resolution_Prover, which itself needs Coinductive and Nested_Multisets_Ordinals — so reproduction goes through the full immutable release, the pattern already used for Deep_Learning and pinned to the same sha256. A POSSIBILITY result, catalogued as the escape boundary dual to the self-knowledge impossibility rows: a process can determine a consistent global state while the computation continues, so the honest obstruction concerns exact knowledge of the PRESENT state and never 'a system cannot know its own global state', which distributed snapshots falsify at useful engineering resolutions. Snapshots escape by relaxing contemporaneity. Path-A reproduction: built at upstream Isabelle, not vendored, no atlas Lean import surface, and no atlas declaration depends on it. Never counts toward headline EXACT/EQUIVALENT coverage. No AI-system reading without a separate reviewed bridge. Evidence: docs/provenance/embedded-self-knowledge-landscape.md.", "original_source_refs": [ "chandy-lamport-1985" ], "related_result_ids": [ "LAND-SELFMEAS-001", "LAND-KNOW-001" ], "relations": [ { "target": "LAND-SELFMEAS-001", "kind": "BOUNDARY_PARTNER", "note": "Not a formal duality: the two do not share a model. This row is the generic whole-state specialization of the knowability kernel, not Breuer's theorem, which is LAND-SELFMEAS-002: an abstract restriction map from global states to apparatus states, with no dynamics, no messages and no algorithm; Chandy-Lamport's is a message-passing distributed system with channels, markers and a recording procedure. What the pair brackets is contemporaneity. The impossibility is about distinguishing the state one is in *now*; the construction recovers a consistent global state by giving up exactly that, recording a cut rather than an instant." } ], "root_import": false, "formalizations": [ { "framework": "Isabelle/HOL", "repository": "https://www.isa-afp.org/entries/Chandy_Lamport.html", "version": "AFP release 2026-02-06 (full release archive, SHA256 b059edd46073479ee8dde45004c2346a7365e5d94cded49d27257cfea66c8879)", "license": "BSD-3-Clause", "reproduced": true, "build_environment": "Isabelle2025-2 (docker makarius/isabelle, digest sha256:9bd33b183c399327c5d554fc8cde27c29b5d2b20cdc6fe7a604caa3f951018fc); full AFP release 2026-02-06", "module": "Snapshot.thy; session Chandy_Lamport", "declarations": [ "snapshot_algorithm_must_terminate", "snapshot_algorithm_is_correct", "Stable_Property_Detection" ], "build_command": "scripts/reproduce_isabelle.sh chandy-lamport" } ], "lean_artifact": null, "public": { "group": "Assembled from other proof assistants", "summary": "A running system can snapshot itself without pausing.", "use": "Why “a system cannot know itself” is too strong.", "title": "Distributed snapshots", "attribution": "Chandy & Lamport" }, "statability": { "verdict": "EXTERNAL_ONLY", "note": "A formalization exists and is reproduced in Isabelle/HOL: AFP Chandy_Lamport, reproduced 2026-08-11 through the full immutable AFP release. No atlas Lean is owed here -- this row records the external artifact, and reproducing it in Lean would be duplication under lean-parsimony.md rather than coverage." } }, { "id": "LAND-SELFMEAS-001", "name": "Self-measurement failure for an embedded observation", "tags": [ "information-theory", "interpretability" ], "result_shape": "POINT_IMPOSSIBILITY", "notes": "Specialization of the LAND-KNOW-001 kernel to the whole-state target. knowable_id_iff_injective says knowing the entire state is exactly injectivity of the observation. not_knowable_state_of_nontrivial_remainder is the load-bearing statement and is axiom-free: when the global state is Read × Rest and Rest can take two values, the pair (r, rest₁), (r, rest₂) collides at Prod.fst while being distinct, so the global state is not knowable from inside; remainderWitness packages the same obstruction as an inspectable certificate. The Breuer PDF is now pinned, but this row remains ungraded against it because it is only the generic fibre skeleton; the source-faithful abstract inference-map core is recorded separately as LAND-SELFMEAS-002 with relationship RELATED. What is mechanized is the information-theoretic skeleton only — non-triviality of the remainder is a hypothesis rather than a conclusion, and there is no measurement dynamics, no time evolution, no quantum case, no apparatus model, and no EPR corollary. This is not to be called Breuer's theorem in any facade, README or status page. No AI-system reading follows without a separate reviewed bridge. original_source_refs is deliberately empty: exactly one row cites the paper, LAND-SELFMEAS-002, so a reader counting Breuer artifacts in the ledger counts one. The relationship to the paper is described here in prose and in the shared provenance note, which is where a cross-reference belongs when there is no grade to justify. Evidence: docs/provenance/self-measurement-kernel.md.", "original_source_refs": [], "related_result_ids": [ "LAND-KNOW-001", "LAND-SELFMEAS-002" ], "relations": [ { "target": "LAND-KNOW-001", "kind": "INSTANTIATES", "note": "The whole-state case: the target is the identity, so knowability is injectivity of the observation." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.knowable_id_iff_injective", "AISafetyAtlas.Knowledge.not_knowable_state_of_nontrivial_remainder", "AISafetyAtlas.Knowledge.remainderWitness" ], "module": "AISafetyAtlas.Knowledge", "atlas_module": "AISafetyAtlas.Knowledge", "build_command": "lake build AISafetyAtlas.Knowledge" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.knowable_id_iff_injective", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.not_knowable_state_of_nontrivial_remainder", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "What an observer can recover", "summary": "To know the whole state, you must be able to tell every pair of states apart.", "use": "The basic version of the limit.", "title": "Whole-state self-measurement", "attribution": "after Breuer" } }, { "id": "LAND-SELFMEAS-002", "name": "Breuer abstract embedded-measurement core", "tags": [ "information-theory", "interpretability" ], "result_shape": "POINT_IMPOSSIBILITY", "notes": "Faithful abstract set-theoretic core of Breuer (1995): §3.2 inference and exact measurement, §3.3 restriction and proper inclusion, §3.5 meshing, and Propositions 1-2. Both propositions are proved, Proposition 1 twice — derived from Proposition 2 and by Breuer's own direct chain — and the two differ in axioms, the derived route depending on none. ProperInclusion is model data rather than a theorem that containment entails non-injectivity; the paper's own footnote 4 gives a contained apparatus without it. The source assumes the restriction surjective; here that is derived from meshing, so the mechanized hypotheses are no stronger than the paper's. Hypotheses and conclusions are both inhabited in AISafetyAtlas.Examples.NonVacuity, so neither proposition is vacuous. Breuer's Corollary is present too, in both its prose form and the printed witness form, so every statement section 3.5 displays is mechanized; sections 4 and 5 display none, they apply these four. Deliberately omitted: physical state-space construction, dynamics, the quantum treatment, EPR, section 3.4's continuity and phase-space-dimension route, and philosophical discussion. Grade EQUIVALENT, not EXACT: the hypotheses are deliberately not the source's, since meshing alone is assumed where the paper assumes meshing and surjectivity, while the Corollary's witness form takes surjectivity back explicitly because it cannot be derived where meshing is what fails. No AI-system or consciousness bridge. Source transcription, the page-image method behind it, and the residual list are in docs/provenance/self-measurement-kernel.md; module design and non-claims are in the module docstring.", "original_source_refs": [ "breuer-1995-self-measurement" ], "related_result_ids": [ "LAND-SELFMEAS-001", "LAND-KNOW-001" ], "relations": [ { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "not_knowable_state_of_properInclusion hands the proper-inclusion pair straight to not_knowable_of_collision." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "relationship": "EQUIVALENT", "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.Embedded.ReadingSet", "AISafetyAtlas.Knowledge.Embedded.ReadingSet.singleton", "AISafetyAtlas.Knowledge.Embedded.ReadingSet.univ", "AISafetyAtlas.Knowledge.Embedded.Restriction", "AISafetyAtlas.Knowledge.Embedded.InferenceMap", "AISafetyAtlas.Knowledge.Embedded.ProperInclusion", "AISafetyAtlas.Knowledge.Embedded.Meshing", "AISafetyAtlas.Knowledge.Embedded.Meshing.restrict_surjective", "AISafetyAtlas.Knowledge.Embedded.ExactlyMeasurable", "AISafetyAtlas.Knowledge.Embedded.MeasuresAllStates", "AISafetyAtlas.Knowledge.Embedded.Distinguishes", "AISafetyAtlas.Knowledge.Embedded.not_knowable_state_of_properInclusion", "AISafetyAtlas.Knowledge.Embedded.no_meshing_inference_distinguishes", "AISafetyAtlas.Knowledge.Embedded.no_meshing_inference_measures_all_states", "AISafetyAtlas.Knowledge.Embedded.exists_state_not_exactly_measurable", "AISafetyAtlas.Knowledge.Embedded.infer_singleton_eq_of_meshing", "AISafetyAtlas.Knowledge.Embedded.exists_singleton_infer_eq_of_infer_eq", "AISafetyAtlas.Knowledge.Embedded.eq_restrict_of_infer_singleton_eq", "AISafetyAtlas.Knowledge.Embedded.no_meshing_inference_measures_all_states_direct", "AISafetyAtlas.Knowledge.Embedded.meshing_fails_of_measuresAllStates", "AISafetyAtlas.Knowledge.Embedded.exists_state_meshing_failure_of_measuresAllStates" ], "module": "AISafetyAtlas.Knowledge.Embedded", "atlas_module": "AISafetyAtlas.Knowledge.Embedded", "build_command": "lake build AISafetyAtlas.Knowledge.Embedded", "scope_delta": { "summary": "Every displayed statement of section 3.5 is mechanized over the paper's own section 3.2 abstraction: Proposition 1 in both the printed existential and the produced negated-universal form, the LEMMA, Proposition 2, and the Corollary in both its prose and printed-witness forms. Sections 4 and 5 add no displayed statement; they apply these four. EQUIVALENT rather than EXACT because the hypotheses are deliberately not the source's: meshing alone is assumed and the section 3.3 surjectivity is derived from it, so the theorems are strictly stronger than printed, while the Corollary's witness form takes surjectivity back as an explicit hypothesis because it cannot be derived where meshing is what fails. Outside section 3.5 the section 3.4 finite-cardinality argument is a separate row (LAND-SELFMEAS-003), its continuity and phase-space-dimension variant is not formalized, and physical apparatus construction, dynamics, the quantum treatment, EPR, and the section 5 universal-validity thesis are omitted.", "evidence": "docs/provenance/self-measurement-kernel.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.not_knowable_state_of_properInclusion", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Meshing.restrict_surjective", "type": "NEW_PROOF", "source_declarations": [], "application": "Abstract embedded measurement: the exact meshing condition entails the source's surjective restriction setup. Breuer (1995), section 3.3 and section 3.5." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.no_meshing_inference_distinguishes", "type": "NEW_PROOF", "source_declarations": [], "application": "Abstract embedded measurement: equal restriction fibres cannot be mutually separated by a meshing inference map. Breuer (1995), Proposition 2." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.no_meshing_inference_measures_all_states", "type": "NEW_PROOF", "source_declarations": [], "application": "Abstract embedded measurement: proper inclusion plus meshing rules out exact measurement of every global state. Breuer (1995), Proposition 1." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.exists_state_not_exactly_measurable", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.infer_singleton_eq_of_meshing", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.exists_singleton_infer_eq_of_infer_eq", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.eq_restrict_of_infer_singleton_eq", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.no_meshing_inference_measures_all_states_direct", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "What an observer can recover", "summary": "A sensor inside a system cannot exactly measure every state of that system.", "use": "The measurement version of the limit.", "title": "Embedded self-measurement", "attribution": "Breuer" } }, { "id": "LAND-SELFMEAS-003", "name": "Physical complement and finite-cardinality bridges to Breuer proper inclusion", "tags": [ "information-theory", "interpretability" ], "result_shape": "CHARACTERIZATION", "notes": "Physical modelling layer over LAND-SELFMEAS-002, not a rewrite of Breuer's section 3.5 core, which continues to take ProperInclusion as model data. Composition derives it from a product or equivalent decomposition with an independently varying remainder, needing no finiteness; Finite derives it from a strict cardinality gap by pigeonhole. The Finite argument is Breuer's own section 3.4 First Attempt, where finitely many apparatus states are said to already exclude exact measurement of all states from inside; it is a bridge rather than the paper's result because section 3.4 is the route Breuer abandons, dropping its continuity variant as difficult to handle and taking the section 3.5 meshing approach instead. That continuity route, which forces equal phase-space dimension in classical mechanics, is not formalized. Composition has no section 3.4 counterpart and is atlas modelling. Both then apply the core no-go under meshing. The boundary has three regions rather than two: a non-injective restriction gives Proposition 1, a bijective one admits a meshing fibre inference map that measures every state, and an injective non-surjective one admits no meshing map at all. Explicit non-claims: not every physical containment has a nontrivial remainder or a cardinality gap, and Bekenstein-type bounds are not encoded as yielding one. Grade RELATED as atlas modelling relative to Breuer, not EXACT. No AI-system bridge. Axiom profiles and the three-region table are in the module docstring; evidence in docs/provenance/self-measurement-kernel.md.", "original_source_refs": [ "breuer-1995-self-measurement" ], "related_result_ids": [ "LAND-SELFMEAS-002", "LAND-SELFMEAS-001", "LAND-KNOW-001" ], "relations": [ { "target": "LAND-SELFMEAS-002", "kind": "BUILDS_ON", "note": "Derives ProperInclusion under product-complement or finite-cardinality hypotheses, then applies the core no-meshing_measures_all theorems." }, { "target": "LAND-SELFMEAS-001", "kind": "REFINES", "note": "Composition restates the product remainder obstruction and feeds it into the Breuer measurement model, not only the Knowable skeleton." }, { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "knowable_whole_state_iff_injective is knowable_id_iff_injective at the apparatus reading." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "relationship": "RELATED", "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.Embedded.Composition.knowable_whole_state_iff_injective", "AISafetyAtlas.Knowledge.Embedded.Composition.properInclusion_iff_not_injective", "AISafetyAtlas.Knowledge.Embedded.Composition.productRestriction", "AISafetyAtlas.Knowledge.Embedded.Composition.equivalentProductRestriction", "AISafetyAtlas.Knowledge.Embedded.Composition.properInclusion_of_nontrivial_remainder", "AISafetyAtlas.Knowledge.Embedded.Composition.not_knowable_productState_of_nontrivial_remainder", "AISafetyAtlas.Knowledge.Embedded.Composition.no_meshing_measures_all_of_nontrivial_remainder", "AISafetyAtlas.Knowledge.Embedded.Composition.properInclusion_of_nontrivial_remainder_of_equiv", "AISafetyAtlas.Knowledge.Embedded.Composition.no_meshing_measures_all_of_nontrivial_remainder_of_equiv", "AISafetyAtlas.Knowledge.Embedded.Composition.fibreInference", "AISafetyAtlas.Knowledge.Embedded.Composition.meshing_fibreInference_of_surjective", "AISafetyAtlas.Knowledge.Embedded.Composition.measuresAll_fibreInference_of_injective", "AISafetyAtlas.Knowledge.Embedded.Composition.meshing_and_measuresAll_fibreInference_of_bijective", "AISafetyAtlas.Knowledge.Embedded.Finite.properInclusion_of_card_lt", "AISafetyAtlas.Knowledge.Embedded.Finite.no_meshing_measures_all_of_card_lt", "AISafetyAtlas.Knowledge.Embedded.Finite.properInclusion_product_of_card_rest_ge_two", "AISafetyAtlas.Knowledge.Embedded.Composition.not_meshing_of_not_surjective" ], "module": "AISafetyAtlas.Knowledge.Embedded.Composition", "atlas_module": "AISafetyAtlas.Knowledge.Embedded.Composition", "build_command": "lake build AISafetyAtlas.Knowledge.Embedded.Composition AISafetyAtlas.Knowledge.Embedded.Finite", "scope_delta": { "summary": "Physical product-complement and finite-cardinality bridges derive ProperInclusion under explicit extra hypotheses. They are not Breuer's abstract core and do not claim that all physical containment forces non-injectivity. Bekenstein is motivation only, not a formal implication.", "evidence": "docs/provenance/self-measurement-kernel.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Composition.properInclusion_of_nontrivial_remainder", "type": "NEW_PROOF", "source_declarations": [], "application": "Physical complement model: independently variable remainder forces restriction collision (product A times R, projection). Atlas modelling layer on Breuer ProperInclusion." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Composition.no_meshing_measures_all_of_nontrivial_remainder", "type": "NEW_PROOF", "source_declarations": [], "application": "Physical complement model plus meshing: nontrivial remainder rules out exact measurement of every product state (Breuer Prop. 1 via derived ProperInclusion)." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Composition.properInclusion_of_nontrivial_remainder_of_equiv", "type": "NEW_PROOF", "source_declarations": [], "application": "Equivalence-general physical complement model: an arbitrary global state representation equivalent to apparatus times remainder inherits the restriction collision." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Composition.no_meshing_measures_all_of_nontrivial_remainder_of_equiv", "type": "NEW_PROOF", "source_declarations": [], "application": "Equivalence-general physical complement model plus meshing: a nontrivial remainder rules out exact measurement of every globally represented state." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Composition.measuresAll_fibreInference_of_injective", "type": "NEW_PROOF", "source_declarations": [], "application": "Positive boundary: injective restriction admits a fibre inference map that exactly measures every global state." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Composition.meshing_and_measuresAll_fibreInference_of_bijective", "type": "NEW_PROOF", "source_declarations": [], "application": "Positive boundary: bijective restriction gives fibre inference that is both meshing and exactly measurable for every global state." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Finite.properInclusion_of_card_lt", "type": "NEW_PROOF", "source_declarations": [], "application": "Finite operational models: strict apparatus/global cardinality gap forces restriction collision (pigeonhole)." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Finite.no_meshing_measures_all_of_card_lt", "type": "NEW_PROOF", "source_declarations": [], "application": "Finite operational models plus meshing: cardinality gap rules out exact measurement of every global state." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Composition.properInclusion_iff_not_injective", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Composition.not_meshing_of_not_surjective", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Composition.meshing_fibreInference_of_surjective", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Embedded.Finite.properInclusion_product_of_card_rest_ge_two", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "What an observer can recover", "summary": "It fails when something else in the system varies, or the sensor is too small.", "use": "Applying the limit without assuming what you want to prove.", "title": "When self-measurement fails", "attribution": "extends Breuer" } }, { "id": "LAND-KNOW-001", "name": "Exact knowability: the observation-factorization kernel", "tags": [ "information-theory", "interpretability" ], "result_shape": "CHARACTERIZATION", "notes": "Generic exact knowability interface. Knowable is defined in decoder form — one rule on observations, uniform in the state — so that the factorization/no-collision characterization stays a theorem rather than definitional unfolding; the same commitment JointObservation.Covers makes, for the same reason. knowable_iff_factorsThrough bridges to Mathlib's Function.FactorsThrough, which supplies the fibrewise construction and the [Nonempty Y] hypothesis; knowable_iff_no_collision is the binder-normalized restatement. IndistinguishabilityWitness is the negative certificate; exists_witness_of_not_knowable is its classical converse. Determines orders observations by informativeness, Knowable.mono transfers knowability upward along it, and not_knowable_comp is the constructive repair boundary — post-processing an unchanged observation cannot create knowability. The factorization content is standard and is Mathlib's; what is packaged here is the decoder-form statement, the witness interface, and monotonicity. No embedded-observer model, no temporal claim, no impossibility claim, and no consciousness result: observation is an arbitrary map, not a projection of a state containing the observer.", "original_source_refs": [], "related_result_ids": [ "LAND-JOINTOBS-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.Knowable", "AISafetyAtlas.Knowledge.IndistinguishabilityWitness", "AISafetyAtlas.Knowledge.Determines", "AISafetyAtlas.Knowledge.knowable_iff_factorsThrough", "AISafetyAtlas.Knowledge.knowable_iff_no_collision", "AISafetyAtlas.Knowledge.not_knowable_of_invariant_transform", "AISafetyAtlas.Knowledge.Determines.refl", "AISafetyAtlas.Knowledge.Determines.trans", "AISafetyAtlas.Knowledge.not_knowable_of_collision", "AISafetyAtlas.Knowledge.not_knowable_of_witness", "AISafetyAtlas.Knowledge.exists_witness_of_not_knowable", "AISafetyAtlas.Knowledge.Knowable.mono", "AISafetyAtlas.Knowledge.not_knowable_comp", "AISafetyAtlas.Knowledge.Check.knowable_congr_observation", "AISafetyAtlas.Knowledge.Check.knowable_congr_property", "AISafetyAtlas.Knowledge.Check.knowable_comp_left_iff" ], "module": "AISafetyAtlas.Knowledge", "atlas_module": "AISafetyAtlas.Knowledge", "build_command": "lake build AISafetyAtlas.Knowledge" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.knowable_iff_factorsThrough", "type": "WRAPPER", "source_declarations": [ "Function.factorsThrough_iff" ] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.knowable_iff_no_collision", "type": "WRAPPER", "source_declarations": [ "Function.factorsThrough_iff" ] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.not_knowable_of_collision", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.not_knowable_of_witness", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Determines.trans", "type": "NEW_PROOF", "source_declarations": [], "application": "Names the composition of two recoveries on the Determines side. It is not a new law: it is defined as Knowable.mono, because Determines and Knowable applied at the same arguments are one proposition. Recorded here rather than on a consumer row because it is a kernel declaration." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.not_knowable_of_invariant_transform", "type": "NEW_PROOF", "source_declarations": [], "application": "The packaged form of not_knowable_of_collision for the case where the colliding pair is produced by an observation-preserving map rather than found. Both Wireheading.ObservationLimits, with the environment complement, and Preference.Knowability, with the anti-rational negation, were building that pair by hand." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.exists_witness_of_not_knowable", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Knowable.mono", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.not_knowable_comp", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "What an observer can recover", "summary": "You can tell things apart only if they look different.", "use": "The shared core behind the observation results.", "title": "Knowability kernel", "attribution": "Workbench infrastructure" }, "escape_routes": [ { "axis": "ADD_INFORMATION", "status": "FORMALIZED", "lean": "AISafetyAtlas.Knowledge.Knowable.mono", "note": "Knowability transfers to any more informative observation: if `finer` Determines `coarser` and the property is knowable from `coarser`, it is knowable from `finer`. The decoder is composed rather than reassembled, so the route is constructive and costs nothing beyond obtaining the finer observation. It was for a long time the only escape route in the ledger the tree proves rather than names; BY-010 now carries two more, each inhabited by a checked model." }, { "axis": "RESTRICT_CLASS", "status": "STATED", "note": "`knowable_iff_no_collision` makes the obstruction exactly a colliding pair, so removing every collision from the quantified domain removes the obstruction by the characterization itself. What it costs is not stated here: nothing says which restrictions are the ones a consumer can actually impose." } ] }, { "id": "LAND-KNOW-DEVICE-001", "name": "Transports between the knowability kernel and inference devices", "tags": [ "information-theory", "oversight" ], "result_shape": "INFRASTRUCTURE", "notes": "Atlas modelling, not a claim of either Wolpert paper. LAND-KNOW-001's Knowable and BY-024's WeaklyInfers are not the same relation - that is a theorem of this development, witnessed both ways on finite models in Examples.Inference.Device, and Inference/Device.lean says so at length. Nothing here retracts it. What this row adds is the conditional half that was missing: the joint object is the device's own setup-and-conclusion pair read as a spine observation (deviceObservation), and an IndistinguishabilityWitness for that pair, lying inside a realized block, stops the block answering the probe. Present in every realized block and straddling the target value (BlockwiseCollision), it refutes Definition 3 and Definition 11 alike, and the physical-knowledge refutation holds for every context W because Definition 11's clause (i) constrains the whole selected block rather than its trace in W. The straddle clause is load-bearing rather than convenient: Examples.Knowledge.Devices exhibits an eight-state device whose every block carries a colliding pair with the target differing, and which weakly infers the target anyway, so the weakened hypothesis would be unsound rather than merely weaker. Travel in the other direction is one lemma: a device answering the probe in every realized block has its conclusion function equal to the probed target, so second projection decodes it. There is deliberately no Knowable-to-WeaklyInfers lemma and no iff between a collision-free block and an answering one; both fail, and the module docstring gives the countermodel for each. Axiom profile measured, not assumed: every refutation and the positive transport depend on no axioms at all, witnessAt reports propext, and the probe-free refutation reports the classical three because constructing a probe is where the choice sits. blockwiseCollision_of_knowable_concl discharges the hypothesis from two conditions an engineer can inspect without mentioning probes - the report factors through the setup, and no reachable configuration separates the value - and Examples.Oversight.Overseer runs both sides of that condition on one four-state system: an overseer whose report reads the world answers every probe while its report stream still does not determine the hazard, and an overseer whose report factors through its configuration is refuted at every configuration, every reconfiguration and every context. The first half is the non-obvious one: satisfying Definition 3 does not make the output decodable, because Definition 3 chooses the configuration per question while a decoder must work in all of them at once. AISafetyAtlas.Knowledge.Check supplies the executable half on the pattern FiniteDecision already sets for coverage - findCollision searches an enumeration for a witness, and the two agreement theorems are what make its output evidence rather than a report, with completeness of the enumeration a hypothesis rather than an instance because it is the whole content of the confirming direction. The lake exe atlas-check reads a finite model as JSON and prints the verdict with the declaration that certifies it; scripts/check_atlas_check.sh asserts that its answers on four shipped models match the Lean proofs in Examples.Oversight.Overseer for the same two devices. The output is what the kernel would give and not a proof term the kernel has checked for that instance, which the tool says itself. A coalition kind decides the joint-observation question over the same certified search, since Covers is definitionally Knowable on the coalition's observation; its shipped models mirror Examples.Oversight.JointObservation.Procurement and reproduce that file's documented failure witness independently. Every verdict reports the worst ambiguity alongside it, so a failing model states how far off it is rather than only that it failed.", "original_source_refs": [], "related_result_ids": [ "LAND-KNOW-001", "BY-024" ], "relations": [ { "target": "LAND-KNOW-001", "kind": "BUILDS_ON", "note": "Consumes IndistinguishabilityWitness as the object crossing the joint and produces Knowable in the positive direction; neither component of the device's pair suffices alone." }, { "target": "BY-024", "kind": "BOUNDARY_PARTNER", "note": "Model delta: Knowable asks one decoder to work at every state, while Definition 3 fixes the conclusion function and lets the choice of setup block vary with the probe. The quantifier alternation is why neither implies the other, and why the transports below are conditional rather than an identification." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.Devices.deviceObservation", "AISafetyAtlas.Knowledge.Devices.BlockAnswers", "AISafetyAtlas.Knowledge.Devices.BlockwiseCollision", "AISafetyAtlas.Knowledge.Devices.not_blockAnswers_of_witness", "AISafetyAtlas.Knowledge.Devices.BlockwiseCollision.not_weaklyInfers", "AISafetyAtlas.Knowledge.Devices.BlockwiseCollision.not_physicallyKnows", "AISafetyAtlas.Knowledge.Devices.blockwiseCollision_of_knowable_concl", "AISafetyAtlas.Knowledge.Devices.knowable_probe_of_forall_blockAnswers", "AISafetyAtlas.Knowledge.Check.findCollision", "AISafetyAtlas.Knowledge.Check.not_knowable_of_findCollision_eq_some", "AISafetyAtlas.Knowledge.Check.knowable_of_findCollision_eq_none" ], "module": "AISafetyAtlas.Knowledge.Devices", "atlas_module": "AISafetyAtlas.Knowledge.Devices", "build_command": "lake build AISafetyAtlas.Knowledge.Devices" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.Devices.not_blockAnswers_of_witness", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Devices.BlockwiseCollision.not_weaklyInfers", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Devices.BlockwiseCollision.not_physicallyKnows", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Devices.knowable_probe_of_forall_blockAnswers", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "What an observer can recover", "summary": "A device blind to the same distinction in every configuration it can reach cannot answer the question, and cannot know the answer either.", "use": "Lets an oversight file apply a device impossibility without restating it, and says exactly when the two notions of recoverability do not connect.", "title": "Knowability and inference devices", "attribution": "Workbench infrastructure" } }, { "id": "LAND-CRMDP-KNOW-001", "name": "True return does not factor through the observed history", "tags": [ "agent-incentives", "information-theory" ], "result_shape": "POINT_IMPOSSIBILITY", "notes": "Reads BY-039's CRMDP complement construction through the LAND-KNOW-001 kernel, with the environment as the unknown. An environment and its complement generate the same observed history but returns summing to the horizon, so a complement pair whose returns differ is a collision and not_knowable_of_collision applies. The primary statement is class-relative: it asks only that some admissible environment class contain such a pair, which is what makes the escape routes exact — restrict the class, add information, relax exactness, or move to a prior. The unrestricted corollary is witnessed at every positive horizon by the zero environment and its complement. The mathematics is not new: Everitt, Krakovna, Orseau, Hutter and Legg supply both ingredients and BY-039 already formalizes them. What this row adds is the factorization reading, the witness certificate, and the import edge making Wireheading a consumer of Knowledge. Quantification is over environments, not over agents: no decoder from observed histories is correct on every environment in the class. Explicit non-claims, including approximation and prior-conditional estimation, are in the module docstring; BY-039's non-claims are inherited.", "original_source_refs": [], "related_result_ids": [ "BY-039", "LAND-KNOW-001", "LAND-AMBIG-001" ], "relations": [ { "target": "LAND-KNOW-001", "kind": "INSTANTIATES", "note": "Takes the state to be the unknown environment, the observation to be the history a fixed policy receives, and the target to be the true finite-horizon return." }, { "target": "BY-039", "kind": "BUILDS_ON", "note": "Both ingredients are BY-039's module: history_complement supplies the collision and return_add_complement supplies the disagreement." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Wireheading.ObservationLimits.observedHistory", "AISafetyAtlas.Wireheading.ObservationLimits.trueReturn", "AISafetyAtlas.Wireheading.ObservationLimits.complementWitness", "AISafetyAtlas.Wireheading.ObservationLimits.not_knowable_trueReturn_of_complement_mem", "AISafetyAtlas.Wireheading.ObservationLimits.not_knowable_trueReturn" ], "module": "AISafetyAtlas.Wireheading.ObservationLimits", "atlas_module": "AISafetyAtlas.Wireheading.ObservationLimits", "build_command": "lake build AISafetyAtlas.Wireheading.ObservationLimits" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Wireheading.ObservationLimits.not_knowable_trueReturn_of_complement_mem", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Wireheading.ObservationLimits.not_knowable_trueReturn", "type": "NEW_PROOF", "source_declarations": [] }, { "atlas_declaration": "AISafetyAtlas.Wireheading.ObservationLimits.returnOver_zeroEnv_complement", "type": "NEW_PROOF", "source_declarations": [] } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "An agent cannot tell how well it is really doing from what it saw.", "use": "Ties reward tampering to what an agent can know.", "title": "Unobservable return", "attribution": "Workbench infrastructure" }, "escape_routes": [ { "axis": "RESTRICT_CLASS", "status": "STATED", "note": "`not_knowable_trueReturn_of_complement_mem` is class-relative by construction: it asks only that some admissible environment class contain an indistinguishable pair whose returns differ. Restricting the class until no such pair survives removes the hypothesis, which is what the module means by the routes being exact. Nothing here proves knowability follows from the restriction; the unrestricted corollary `not_knowable_trueReturn` is what shows the full class is sufficient, not necessary." }, { "axis": "ADD_INFORMATION", "status": "NAMED_ONLY", "note": "Named as a route in the module docstring. The kernel's `Knowable.mono` is the general form and would apply to any observation refining the observed history, but no declaration instantiates it here, so what a richer observation would have to reveal is undetermined." }, { "axis": "RELAX_EXACTNESS", "status": "NAMED_ONLY", "note": "Named as a route in the module docstring. `Knowable` is exact recovery; the row's explicit non-claims say approximation is not addressed, so no bound on how well the return can be approximated is asserted either way." }, { "axis": "MOVE_TO_PRIOR", "status": "NAMED_ONLY", "note": "Named as a route in the module docstring. The statement quantifies over environments in a class rather than averaging over a prior; the row's explicit non-claims exclude prior-conditional estimation, so nothing is claimed about the Bayesian question." } ] }, { "id": "LAND-CRMDP-GRID-001", "name": "Everitt et al. Theorem 11 over the source's own uniform reward grid", "tags": [ "agent-incentives" ], "notes": "Everitt, Krakovna, Orseau, Hutter and Legg, arXiv:1705.08417v2, Theorem 11, over the hypothesis class the theorem actually names. WHAT THIS ADDS TO BY-039. AISafetyAtlas.Wireheading.CRMDP proves the conclusion for a class whose rewards range over the whole of Set.Icc 0 1, with the three extrema as structure fields. Theorem 11's own class is the full product of functions into a UNIFORM discretisation R = {r_1, ..., r_n} of [0,1] with r_1 = 0 and r_n = 1, and that class is a proper subclass of the interval-valued one which the atlas had no type for - so Theorem 11 at print's own hypotheses was not available. GridEnv is that type, at n = m + 2 so that n >= 2 is forced rather than assumed. WHY UNIFORMITY IS LOAD-BEARING: it is exactly what makes x -> 1 - x map the grid to itself, so complementing a member of the class lands back in the class; gridVal_gridRev is the fact, and gridVal_strictMono, gridVal_zero and gridVal_last are the rest of the printed sentence. A non-uniform discretisation of [0,1] containing 0 and 1 need not have the property, and then 'the classes contain all functions' would not supply the complemented environment. EXTREMA ARE DERIVED, NOT ASSUMED. This is the item BY-039's own non-claims listed as open. Print takes max and min over policies in its equations (4) and (5) without comment, which is legitimate because its state, action and reward sets are finite; the argument is spelled out here. At horizon t a return reads a policy only at the t histories the rollout reaches; those histories have pairwise distinct action-list lengths (historyUpTo_length), so every length-t action sequence is realised by seqPolicy and every policy's return is a sequence's return (stateAt_eq_seqState, stateAt_seqPolicy); a rollout of length k reads only the first k actions (seqState_congr); hence the return factors through the finite type Fin t -> Action and exists_max_gridReturn and exists_min_gridReturn attain the extrema over the WHOLE policy function space. Extrema over environments come from Fintype (GridEnv m State), which is print's finite class. toComplementedClass therefore discharges every field of Corruption.ComplementedClass rather than supplying any, and everitt_theorem_eleven_gridClass is Theorem 11 over print's class. NO DUPLICATION OF THE RUN LAYER: run, stateAt, historyUpTo and returnOver are CRMDP's, reached through toEnv, so there is one return function in the tree and not two. The price is that toEnv's corruption field is defined on the whole interval where print's C is defined on the grid; off the grid it is the identity, and toEnv_corruption_gridVal shows the choice is never observed, because a channel is only ever evaluated at a true reward and every true reward is on the grid. run_congr_observed - a run depends on an environment only through its observed reward function - is what makes the embedding a modelling convenience rather than a second development. STILL NOT PRINT'S, and this is now the only axis: the transition is DETERMINISTIC and the return is a bare sum, where print's T is a stochastic kernel and its returns are expectations. Policies are deterministic too, where print's are 'possibly stochastic'; the maximum of an affine functional over randomisations of finitely many deterministic policies is attained at a deterministic one, so the printed max should agree, and everitt_theorem_eleven_fullMixedClass now carries the stochastic-transition, stochastic-policy class; everitt_theorem_eleven_gridClass itself remains an extremum over deterministic policies only. One transition per class, as in CRMDP. AISafetyAtlas.Examples.Wireheading.RewardGrid works the smallest instance of print's setting - n = 2, two states, two actions, horizon 1 - and proves the maximal worst-case regret there is at least 1, so the half-maximal bound is a statement about a nonzero quantity rather than 0/2 <= 0. Supporting wireheading mathematics, not itself a Table-1 row.", "original_source_refs": [ "survey-ref-077" ], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "relationship": "RELATED", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "RewardGrid.everitt_theorem_eleven_gridClass", "RewardGrid.toComplementedClass", "RewardGrid.exists_max_gridReturn", "RewardGrid.exists_min_gridReturn", "RewardGrid.gridVal_gridRev", "RewardGrid.gridVal_strictMono", "RewardGrid.gridReturn_add_complement", "RewardGrid.run_congr_observed", "RewardGrid.halfMaximalRegretBound" ], "module": "AISafetyAtlas.Wireheading.RewardGrid", "build_command": "lake build AISafetyAtlas.Wireheading.RewardGrid AISafetyAtlas.Examples.Wireheading.RewardGrid", "scope_delta": { "summary": "Print's reward grid and print's finiteness, with the extrema derived. What is left between this and Theorem 11 is the stochastic layer: `transition` is a function rather than a kernel, `returnOver` is a sum rather than an expectation, and the policies whose extrema are attained are the deterministic ones.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Wireheading.RewardGrid.everitt_theorem_eleven_gridClass", "type": "NEW_PROOF", "source_declarations": [ "RewardGrid.everitt_theorem_eleven_gridClass", "RewardGrid.toComplementedClass", "RewardGrid.exists_max_gridReturn", "RewardGrid.exists_min_gridReturn", "RewardGrid.gridVal_gridRev", "RewardGrid.gridVal_strictMono", "RewardGrid.gridReturn_add_complement", "RewardGrid.run_congr_observed", "RewardGrid.halfMaximalRegretBound" ], "application": "Reward corruption: over the full class of true-reward and corruption functions valued in a uniform discretisation of the unit interval, with deterministic transitions, one transition per class and deterministic policies (everitt_theorem_eleven_fullMixedClass covers the stochastic class), every policy suffers at least half the worst-case regret of a worst policy - with the optimal and worst policies and the worst environment all shown to exist rather than assumed. Theorem 11 of Everitt, Krakovna, Orseau, Hutter and Legg, arXiv:1705.08417v2." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "With a corruptible reward channel, no policy can be much better than the worst one.", "use": "Bounding what any agent can achieve when its reward signal can lie.", "title": "Corrupt-reward no free lunch, on the source's own grid", "attribution": "Everitt, Krakovna, Orseau, Hutter & Legg" } }, { "id": "LAND-FANO-001", "name": "Fano's inequality at both printed constants, and its sharpness", "tags": [ "information-theory" ], "notes": "Cover and Thomas, Elements of Information Theory (2nd ed.), section 2.10, mechanized against the printed statement. fano_of_log_le is the master form: it takes any uniform bound L on log |A \\ {X' w}| and concludes H[X | X'] <= P(err) * L + binEntropy(P(err)). Both printed constants are corollaries of that one lemma — fano at the sharp log(|A| - 1) of (2.140), fano_unrestricted at the log |A| of (2.139) — where the chapter argues them separately. fano_of_embedding carries the estimate into a different type through an injection, which the printed statement does not do; the printed corollary for an estimator X-hat : Y -> X is fano_of_estimator_chain in the examples. entropy_le_fano is the no-observation remark with its sharp -1. Sharpness is entropy_eq_fano_of_witness in AISafetyAtlas.Examples.InformationTheory.Fano, which attains equality at every p in [0,1] rather than at a single point, matching the one-parameter family the chapter prints. Alphabet membership is a pointwise hypothesis (forall w, X w in A), which is what the printed typing gives; an almost-everywhere version would exceed print and is not claimed. Coverage and scope are graded row by row in docs/provenance/source-coverage-audit.md.", "original_source_refs": [ "cover-thomas-2006" ], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; PFR 7d6404b79b11b1a89dd1b6f997a10b14c208c4ed; atlas IN_TREE", "declarations": [ "fano_of_log_le", "fano", "fano_unrestricted", "fano_of_embedding", "entropy_le_fano", "le_errorProb", "fano_le_log_two_add", "condEntropy_eq_error_split", "condEntropy_le_error_bound" ], "module": "AISafetyAtlas.InformationTheory.Fano", "build_command": "lake build AISafetyAtlas.InformationTheory.Fano AISafetyAtlas.Examples.InformationTheory.Fano" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.InformationTheory.fano_of_log_le", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.condEntropy_eq_error_split", "AISafetyAtlas.InformationTheory.condEntropy_le_error_bound", "AISafetyAtlas.InformationTheory.entropy_errorIndicator" ], "application": "Estimation limits: the conditional entropy left after an estimate is bounded by the error probability times the log of the alphabet, plus one bit. Cover and Thomas Theorem 2.10.1, with both printed constants as instances of a single lemma." }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.fano", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.fano_of_log_le" ] }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.fano_unrestricted", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.fano_of_log_le" ] }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.fano_of_embedding", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.fano_of_log_le" ] }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.entropy_le_fano", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.fano_of_log_le" ] } ] }, "public": { "group": "What an observer can recover", "summary": "Guess wrong often enough and the thing you are guessing about stays uncertain — by a fixed amount you can compute.", "use": "Turning an error rate into a hard floor on what is still unknown.", "title": "Fano's inequality", "attribution": "Fano; Cover & Thomas" } }, { "id": "LAND-OVERSIGHT-VARIETY-001", "name": "Seeing and doing are independent oversight capacities", "tags": [ "oversight", "control-theory" ], "notes": "A bridge module over an oversight model, built on Ashby's counting bound (BY-004) and the knowability kernel. Sit is the set of situations an overseer is answerable for, Obs what it observes, Act its interventions, Out the outcomes; an overseer is a map Obs -> Act, so the policy is observation-mediated rather than omniscient, which is the conservative direction. Forces says the composed policy pins the outcome at a single target in EVERY situation, which is the strongest reading of 'oversight works' and therefore the reading whose necessary conditions are weakest. not_forces_of_card_lt: with fewer interventions than situations, and an effect table in which a fixed intervention still distinguishes situations, no policy forces the outcome - and the hypothesis list contains nothing about what the overseer observes, which is the theorem rather than an omission in it. forces_of_constant_effect is the other corner and the boundary of the first: an intervention whose effect is constant forces the target from a blind policy, and a constant column is exactly where the structural hypothesis fails. Examples.Oversight.VarietyBound exhibits both corners at three situations each and states them as one conjunction, so the independence is a checked object rather than prose. Taken alone the first theorem would be close to renaming Ashby's regulator an overseer, which methodology.md says does not earn a bridge grade; what is offered as more than a rename is the pair, a statement about the relation between two capacities that the oversight and control clusters formalize separately and neither states alone. A reviewer who rejects that reading should reject the bridge grade; the declarations are correct either way. Review package at docs/interpretation-reviews/review-oversight-varietybound.md. Accepted at REVIEWED by the maintainer on 2026-08-17, statement and scoped interpretation both, as conditionals and licensing none of the package's forbidden claims. The decision is not recorded in this row: ai_interpretation_status and interpretation_review are rejected on rows without an informal_claim, which is every LAND- row, so the generated reviewed-bridge count reads 2 and understates it. The package records the decision and the three ways to make it recordable.", "original_source_refs": [], "related_result_ids": [ "BY-004", "LAND-JOINTOBS-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Forces", "not_forces_of_card_lt", "forces_of_constant_effect", "forces_of_constant_effect_of_not_knowable", "cannotForce", "not_forces_of_cannotForce", "exists_cannotForce_false_and_forces" ], "module": "AISafetyAtlas.Oversight.VarietyBound", "build_command": "lake build AISafetyAtlas.Oversight.VarietyBound AISafetyAtlas.Oversight.VarietyCheck AISafetyAtlas.Examples.Oversight.VarietyBound" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Oversight.not_forces_of_cannotForce", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Oversight.cannotForce", "AISafetyAtlas.Oversight.not_forces_of_card_lt" ], "application": "The agreement theorem behind atlas-check's variety kind: a true verdict from the executable checker rules out every policy over every observation type, so a consumer who cannot read Lean still gets the quantifier the theorem has." }, { "atlas_declaration": "AISafetyAtlas.Oversight.exists_cannotForce_false_and_forces", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Oversight.cannotForce", "AISafetyAtlas.Oversight.Forces" ], "application": "Why a false verdict is not a clearance: a table where the counting bound is silent and forcing succeeds, so 'not obstructed' can never be read as 'not possible'." }, { "atlas_declaration": "AISafetyAtlas.Oversight.forces_of_constant_effect", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Oversight.Forces" ], "application": "Oversight: control without observation. The boundary of the counting bound, since a constant column is exactly where its structural hypothesis fails.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-forces-of-constant-effect.md" } }, { "atlas_declaration": "AISafetyAtlas.Oversight.forces_of_constant_effect_of_not_knowable", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Oversight.forces_of_constant_effect", "AISafetyAtlas.Knowledge.Knowable" ], "application": "Oversight: the two capacities stated in one place - unknowable hazard, forced outcome.", "review_status": "STATEMENT_REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": false, "evidence": "docs/interpretation-reviews/review-forces-of-constant-effect-of-not-knowable.md" } } ] }, "public": { "group": "Limits of control and regulation", "summary": "An overseer that can tell every situation apart may still be unable to steer the outcome, and one that can tell nothing apart may be able to. What you can see and what you can do are separate budgets.", "use": "Checking whether an oversight proposal is short of information, short of levers, or both.", "title": "Seeing is not steering", "attribution": "Atlas, over Ashby" } }, { "id": "LAND-KNOWENTROPY-001", "name": "Knowability measured: zero conditional entropy, and Fano's floor on every decoder", "tags": [ "information-theory", "interpretability" ], "notes": "Joins the knowability kernel to the entropy layer, in the one direction that holds exactly. condEntropy_eq_zero_of_knowable: a decoder costs no entropy, proved from InformationTheory.condEntropy_comp_self_left. not_knowable_of_condEntropy_ne_zero is the contrapositive and the usable direction - an estimated entropy certifies unknowability without ever naming the colliding pair, where IndistinguishabilityWitness requires exhibiting one. le_errorProb_of_decoder is the quantitative form: EVERY decoder built from the observation, not merely some decoder, errs with probability at least (H[property | observation] - log 2) / log |A|. It is reached by instantiating Fano's converse (InformationTheory.le_errorProb, Cover and Thomas 2.132) at the decoder's output and then transferring the bound back to the observation along data processing, since a decoder is a coarsening and isMarkovChain_comp plus condEntropy_le_condEntropy_of_isMarkovChain say coarsening never lowers conditional entropy. No new information-theoretic content is claimed; the constant is Cover and Thomas's. The converse fails and the failure is exhibited rather than asserted: Examples.Knowledge.Entropy carries a property that is not Knowable whose conditional entropy is zero, by putting a point mass on one of two colliding states. Entropy answers a question about the measure and Knowable answers a question about every state, so entropy is a sound certificate of unknowability and an unsound certificate of knowability. No AI-system reading is asserted: observation is an arbitrary measurable map.", "original_source_refs": [ "cover-thomas-2006" ], "related_result_ids": [ "LAND-FANO-001", "LAND-DPI-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; PFR 7d6404b79b11b1a89dd1b6f997a10b14c208c4ed; atlas IN_TREE", "declarations": [ "condEntropy_eq_zero_of_knowable", "not_knowable_of_condEntropy_ne_zero", "le_errorProb_of_decoder" ], "module": "AISafetyAtlas.Knowledge.Entropy", "build_command": "lake build AISafetyAtlas.Knowledge.Entropy AISafetyAtlas.Examples.Knowledge.Entropy" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.condEntropy_eq_zero_of_knowable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Knowledge.Knowable", "AISafetyAtlas.InformationTheory.condEntropy_comp_self_left" ], "application": "A property recoverable from an observation leaves that observation no uncertainty about it. The entropy side of the knowability kernel." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.not_knowable_of_condEntropy_ne_zero", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Knowledge.condEntropy_eq_zero_of_knowable" ], "application": "A positive conditional entropy certifies that no decoder exists, without exhibiting the pair the observation confuses." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.le_errorProb_of_decoder", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.le_errorProb", "AISafetyAtlas.InformationTheory.condEntropy_le_condEntropy_of_isMarkovChain", "AISafetyAtlas.InformationTheory.isMarkovChain_comp" ], "application": "How badly unknowability bites: a floor on the error rate of every decoder built from the observation, uniform over decoders." } ] }, "public": { "group": "What an observer can recover", "summary": "If what you want to know is not fixed by what you can see, no procedure reading your observations gets it right more often than a computable floor allows.", "use": "Turning \"this monitor cannot distinguish those cases\" into a rate no amount of cleverness downstream improves on.", "title": "Unknowability has an error rate", "attribution": "Atlas, over Fano; Cover & Thomas" } }, { "id": "LAND-DPI-001", "name": "The data-processing inequality, its equality case, and the conditioning counterexamples", "tags": [ "information-theory" ], "notes": "Cover and Thomas, Elements of Information Theory (2nd ed.), section 2.8. mutualInfo_le_of_isMarkovChain is Theorem 2.8.1 by the printed argument, two expansions of I(X ; Y, Z). Two statements the chapter asserts inside that proof but does not prove are proved here: mutualInfo_eq_iff_isMarkovChain is the equality case, and mutualInfo_le_of_isMarkovChain' is the dual the text leaves to the reader. isMarkovChain_comp is wider than the printed corollary, taking any measurable g rather than a named function. Markov chains are *defined* as CondIndepFun, which is (2.118)'s right-hand side where the book starts from the factorization (2.117). isMarkovChain_iff_measure_factorizes proves the two equivalent, so the choice of definition is no longer a convention the section trades on. Both directions run at the printed strength. Forward, measure_factorizes_of_isMarkovChain gives p(y)*p(x,y,z) = p(x,y)*p(y,z) at the printed point masses and in fact at every measurable s, t. Backward, isMarkovChain_iff_measure_factorizes_singleton takes the hypothesis at point masses alone, carried up to measurable sets by measure_preimage_inter_eq_tsum - a measurable slice of a countable-valued variable splits over its point masses - applied once in X and once in Z. The set-level isMarkovChain_iff_measure_factorizes is the wider conclusion and needs only Y measurable, with no countability or measurable singletons on S and U. Both printed conditioning counterexamples are in AISafetyAtlas.Examples.InformationTheory.DataProcessing: the chapter's integer sum at half a bit, and an XOR pair at a full bit.", "original_source_refs": [ "cover-thomas-2006" ], "related_result_ids": [ "LAND-FANO-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; PFR 7d6404b79b11b1a89dd1b6f997a10b14c208c4ed; atlas IN_TREE", "declarations": [ "isMarkovChain_iff_measure_factorizes_singleton", "measure_preimage_inter_eq_tsum", "isMarkovChain_iff_measure_factorizes", "measure_factorizes_of_isMarkovChain", "mutualInfo_le_of_isMarkovChain", "mutualInfo_le_of_isMarkovChain'", "mutualInfo_eq_iff_isMarkovChain", "mutualInfo_comp_le", "condMutualInfo_le_mutualInfo", "isMarkovChain_comp", "IsMarkovChain.symm", "condEntropy_le_condEntropy_of_isMarkovChain" ], "module": "AISafetyAtlas.InformationTheory.DataProcessing", "build_command": "lake build AISafetyAtlas.InformationTheory.DataProcessing AISafetyAtlas.Examples.InformationTheory.DataProcessing" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.InformationTheory.isMarkovChain_iff_measure_factorizes_singleton", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.isMarkovChain_iff_measure_factorizes", "AISafetyAtlas.InformationTheory.measure_preimage_inter_eq_tsum" ], "application": "Markov chains: the book's defining factorization at point masses and conditional independence are the same condition. Cover and Thomas (2.118), derived at the printed hypothesis in both directions." }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.isMarkovChain_iff_measure_factorizes", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.IsMarkovChain" ], "application": "Markov chains: the book's defining factorization (2.117) and conditional independence are the same condition. Cover and Thomas (2.118), derived rather than adopted as a convention." }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.measure_factorizes_of_isMarkovChain", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.isMarkovChain_iff_measure_factorizes" ] }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.mutualInfo_le_of_isMarkovChain", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.mutualInfo_chain_rule", "AISafetyAtlas.InformationTheory.mutualInfo_sub_eq" ], "application": "Post-processing cannot create information: along a Markov chain X -> Y -> Z, no function of Y tells you more about X than Y does. Cover and Thomas Theorem 2.8.1." }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.mutualInfo_eq_iff_isMarkovChain", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.mutualInfo_sub_eq", "AISafetyAtlas.InformationTheory.isMarkovChain_iff_condMutualInfo_eq_zero" ], "application": "The equality case, which the chapter asserts inside the proof of 2.8.1 without proving it." }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.mutualInfo_comp_le", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.mutualInfo_le_of_isMarkovChain", "AISafetyAtlas.InformationTheory.isMarkovChain_comp" ] }, { "atlas_declaration": "AISafetyAtlas.InformationTheory.condMutualInfo_le_mutualInfo", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.InformationTheory.mutualInfo_chain_rule'" ] } ] }, "public": { "group": "What an observer can recover", "summary": "No amount of processing a measurement adds information about what was measured.", "use": "Why a pipeline cannot recover what its first stage discarded.", "title": "Data processing", "attribution": "Cover & Thomas" } }, { "id": "LAND-CAUSAL-PEARLCBN-001", "name": "Pearl causal Bayesian networks as a condition on an interventional family", "tags": [ "interpretability" ], "result_shape": "INFRASTRUCTURE", "notes": "GROUNDING. Pearl, Causality: Models, Reasoning, and Inference, 2nd ed., 2009, Definition 1.3.1 and equation (1.37), graded in docs/provenance/source-coverage-audit.md section 7. CONTENT. Definition 1.3.1 is a condition on a FAMILY of interventional distributions, so the family is the object: InterventionalFamily is P-star at the definition's own index set, hard interventions, and a statement about it grades against the definition rather than past it. IsCausalBayesNetwork is the four printed conjuncts, with one observational mechanism family q hoisted once and each member factorized by its own r. eq_family_of_isCausalBayesNetwork DERIVES the truncated product rather than assuming it, which is what print does. Model.isCausalBayesNetwork instantiates the condition at the atlas kernel. A READING, RECORDED. Print writes condition (iii) as a quotient of conditional probabilities; the atlas states conditions (ii) and (iii) on TABLES. The two differ exactly on parent configurations of probability zero, where the quotient is undefined and table equality still binds. That is deliberate: null parent fibres are reachable, which is why Causal.MarginClass carries margin conditions at all, so a quotient rendering would go vacuous precisely where a degenerate mechanism needs constraining. The audit's 'Readings that are not transcriptions' section lists it. WHY THIS ROW EXISTS SEPARATELY. Causal.BayesianNetwork was hosted by no registry row until 2026-08-22, so section 7 graded it while every generated status page omitted it. RESULT. RELATED, not EXACT: the row records that the atlas states Pearl's definition and derives his equation, not that a printed theorem was reproduced.", "original_source_refs": [ "pearl-2009-causality" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Causal.InterventionalFamily", "AISafetyAtlas.Causal.ConditionalTables", "AISafetyAtlas.Causal.ConditionalTables.family", "AISafetyAtlas.Causal.forcedTable", "AISafetyAtlas.Causal.IsCausalBayesNetwork", "AISafetyAtlas.Causal.eq_family_of_isCausalBayesNetwork", "AISafetyAtlas.Causal.isCausalBayesNetwork_family", "AISafetyAtlas.Causal.isCausalBayesNetwork_iff", "AISafetyAtlas.Causal.Model.tables", "AISafetyAtlas.Causal.Model.interventionalFamily", "AISafetyAtlas.Causal.Model.interventionalFamily_eq", "AISafetyAtlas.Causal.Model.isCausalBayesNetwork" ], "relationship": "RELATED", "scope_delta": { "summary": "Scope Same against Definition 1.3.1: the four printed conjuncts, at Pearl's own index set of hard interventions. Conditions (ii) and (iii) are stated on tables where print writes a quotient of conditional probabilities, which is a reading and not a scope change -- the two differ only on parent configurations of probability zero, where print's quotient is undefined and the atlas condition still binds, and the atlas takes the stronger side deliberately because null parent fibres are reachable in this kernel. Equation (1.37) is derived, not assumed.", "evidence": "docs/provenance/source-coverage-audit.md" }, "module": "AISafetyAtlas.Causal.BayesianNetwork", "atlas_module": "AISafetyAtlas.Causal.BayesianNetwork", "build_command": "lake build AISafetyAtlas.Causal.BayesianNetwork AISafetyAtlas.Examples.Causal.BayesianNetwork" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Causal.eq_family_of_isCausalBayesNetwork", "type": "NEW_PROOF", "source_declarations": [], "application": "Pearl's equation (1.37) as a consequence of Definition 1.3.1 rather than a separate axiom: on the forced coordinates condition (ii) replaces the mechanism by a point mass, on the free ones condition (iii) replaces it by the observational mechanism, and condition (i) multiplies them. Print derives it the same way, so deriving it is what makes this the printed definition and not a restatement of its corollary." } ] }, "public": { "group": "What an observer can recover", "summary": "Pearl's causal Bayesian network, stated as a condition on a whole family of interventional distributions rather than on one distribution, with the truncated-product formula derived from it.", "use": "Deciding whether a family of interventional distributions is generated by a single causal graph, and obtaining the truncated product when it is." } }, { "id": "LAND-CAUSAL-DECISIONNET-001", "name": "Decision tasks as causal influence diagrams, with the decision and the utility as vertices", "tags": [ "interpretability", "agent-incentives" ], "result_shape": "INFRASTRUCTURE", "notes": "GROUNDING. Richens and Everitt, Robust agents learn causal world models, ICLR 2024, Section 2.2 and Appendix A.1, read from the published proceedings PDF richens-everitt-2024-iclr-proceedings.pdf, sha256 143d458cbee2f4f5d04d7380f6741e8105965d1ee94834e1ddb3bd231722a0e7. Graded in docs/provenance/source-coverage-audit.md section 6. CONTENT. DecisionNetwork is Definition 4: a single-decision single-utility CID, which print defines as a CBN whose variables are partitioned into decision, utility and chance, so it is built on Causal.Model rather than on the structural layer. DecisionPolicy is print's pi(d | pa_D). Model.withPolicy is do(D = pi(pa_D)) as a SOFT intervention: the parent set is untouched and withPolicy_parents records it, because a policy READS pa_D and RE24's own Definition 2 severs a hard-intervened vertex's incoming edges. Reading print's do as a hard intervention here would delete the observations the policy is defined on. expectedUtility, IsOptimal and regret are Section 2.2's three sentences at print's own quantifiers. IsUnmediated is Assumption 1, with proper ancestors and descendants as print's notation section requires - print writes 'Anc_i and Desc_i refer to proper ancestors and descendants', and on print's own Figure 1 the decision is a parent of the utility, so an improper reading would make Assumption 1 unsatisfiable. mem_parents_utility_of_isUnmediated is the graph half of Appendix A.1 Lemma 1(iii): 'D in Anc_U which with Desc_D and Anc_U disjoint implies D in Pa_U'. exists_utilityFunction recovers print's U(pa_U) as an actual function of the parents from IsDeterministicUtility. RESULT. Assumption 1 is INHABITED, not merely defined: figCID is print's own Figure 1 training diagram, figIsUnmediated discharges Assumption 1 on it, and figDecision_mem_parents_utility runs the Appendix step. Without that witness both Assumption-1 theorems would be vacuous. SCOPE. Same against Definition 4, Assumption 1 and the Lemma 1(iii) graph step. ONE DISCLOSED WIDENING: print says the utility variable is a real-valued function of its parents, and DecisionNetwork does not make that a field, so expectedUtility is defined on a strictly larger class. IsDeterministicUtility pins print's case and exists_utilityFunction is the lemma that recovers print's function there. CLOSED THROUGH THIS ROW, 2026-09-20. The Section 2.2 rows on Causal.Decision were Narrower and are Same. Those declarations are the Assumption-1 projection with the decision and the utility OUTSIDE the graph, and expectedUtility_eq_value now proves that projection agrees with expectedUtility on a diagram satisfying Assumption 1 (with the Appendix A.2 normalization), with optimalValue_eq_expectedUtility for the optima. That agreement theorem closed them: section 6 of the coverage audit grades the Section 2.2 rows Same since 2026-09-20. Lemma 1 itself stays uncovered: its clauses (i) and (ii) and the first half of (iii) run through Assumption 2, domain dependence, which is not formalized. INHERITED. A DecisionNetwork is a Causal.Model, so it carries that structure's finite vertex set, parents as a Finset and N-ranked acyclicity; those are graded on Causal.Model's own rows.", "original_source_refs": [ "richens-everitt-2024" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Causal.DecisionNetwork", "AISafetyAtlas.Causal.DecisionPolicy", "AISafetyAtlas.Causal.Model.withPolicy", "AISafetyAtlas.Causal.Model.withPolicy_parents", "AISafetyAtlas.Causal.Model.properAncestors", "AISafetyAtlas.Causal.Model.properDescendants", "AISafetyAtlas.Causal.Model.cpt_eq_one_unique", "AISafetyAtlas.Causal.DecisionNetwork.expectedUtility", "AISafetyAtlas.Causal.DecisionNetwork.IsOptimal", "AISafetyAtlas.Causal.DecisionNetwork.regret", "AISafetyAtlas.Causal.DecisionNetwork.regret_nonneg", "AISafetyAtlas.Causal.DecisionNetwork.IsUnmediated", "AISafetyAtlas.Causal.DecisionNetwork.IsDeterministicUtility", "AISafetyAtlas.Causal.DecisionNetwork.exists_utilityFunction", "AISafetyAtlas.Causal.DecisionNetwork.decision_notMem_parents_of_isUnmediated", "AISafetyAtlas.Causal.DecisionNetwork.mem_parents_utility_of_isUnmediated", "AISafetyAtlas.Examples.Causal.DecisionNetwork.figCID", "AISafetyAtlas.Examples.Causal.DecisionNetwork.figIsUnmediated", "AISafetyAtlas.Examples.Causal.DecisionNetwork.figDecision_mem_parents_utility", "AISafetyAtlas.Examples.Causal.DecisionNetwork.figIsDeterministicUtility" ], "relationship": "RELATED", "scope_delta": { "summary": "Scope Same against RE24 Definition 4, Assumption 1 and the graph half of Appendix A.1 Lemma 1(iii). One disclosed widening: print's clause that the utility is a real-valued function of its parents is carried as the hypothesis IsDeterministicUtility rather than as a field, so expectedUtility is defined on a larger class than print's; exists_utilityFunction recovers print's U(pa_U) where the hypothesis holds. do(D = pi(pa_D)) is rendered as a soft intervention, keeping pa_D, because a policy reads its observations. The Section 2.2 rows on Causal.Decision are the Assumption-1 projection, and expectedUtility_eq_value ties them to expectedUtility: on an unmediated diagram, with the Appendix A.2 normalization, the projection's Model.value equals expectedUtility, and optimalValue_eq_expectedUtility identifies the two optima. That closed them: section 6 of the coverage audit grades the Section 2.2 expected-utility and regret rows Same since 2026-09-20.", "evidence": "docs/provenance/source-coverage-audit.md" }, "module": "AISafetyAtlas.Causal.DecisionNetwork", "atlas_module": "AISafetyAtlas.Causal.DecisionNetwork", "build_command": "lake build AISafetyAtlas.Causal.DecisionNetwork AISafetyAtlas.Examples.Causal.DecisionNetwork" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Causal.DecisionNetwork.mem_parents_utility_of_isUnmediated", "type": "NEW_PROOF", "source_declarations": [], "application": "The graph half of RE24 Appendix A.1 Lemma 1(iii). Print compresses it to one clause -- D in Anc_U which with Desc_D and Anc_U disjoint implies D in Pa_U -- and the proof is the expansion: if the decision reached the utility through any intermediate vertex, that vertex would be a proper descendant of the decision and a proper ancestor of the utility at once, which Assumption 1 forbids. Formally the ancestor closure of the utility, minus the decision and its proper descendants, is parent-closed and contains the utility, so it contains the whole closure - and therefore the decision, which it does not." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.DecisionNetwork.figIsUnmediated", "type": "NEW_PROOF", "source_declarations": [], "application": "Assumption 1 discharged on print's own Figure 1 training diagram. Without an inhabitant every theorem conditional on IsUnmediated would be vacuously true, including the Lemma 1(iii) step above." } ] }, "public": { "group": "What an observer can recover", "summary": "A decision task written as a causal influence diagram with the decision and the utility as vertices of the network, together with expected utility, optimality and regret at the source's own quantifiers, and the unmediated-task assumption with the graph consequence the source draws from it.", "use": "Stating a single-decision single-utility decision task where the decision node's table is the policy, and deriving that under an unmediated task the decision's only route to the utility is the direct edge.", "title": "Decision tasks as influence diagrams", "attribution": "Richens and Everitt, ICLR 2024, Section 2.2 and Appendix A.1" } }, { "id": "LAND-CAUSAL-COLLISION-001", "name": "Margins do not imply behavioral identifiability", "tags": [ "interpretability", "agent-incentives" ], "notes": "GROUNDING. Layer 0 is a finite categorical CBN construction over an ordered field: RE24 Appendix A.2 supplies dimensions and full-simplex CPTs; RE24 Definition 2 and equation (1) supply arbitrary local state maps and their preimage-sum factors; Pearl 2009 equation (1.37) supplies the hard-intervention product reading, certified by the fixProfile Dirac and evaluation theorems. Model is constructive (G, theta), not Pearl Definition 1.3.1. Model, Skeleton and the mixture spaces carry a value field, so probability mixtures are the RE24 Definition-3 simplex over whichever field a statement picks rather than a rational specialization of it; witnesses are computed at Q and Skeleton.marginClass_mapRat and Skeleton.behaviorEq_mapRat transport them into any characteristic-zero ordered field. Layer 1 is the unmediated Assumption-1 projection of finite policies and utility, not a CID. The six margin conditions and masked transform family specialize to the printed binary A2 composite; their categorical forms are atlas extensions, not RE24 or Uhler et al. definitions. RESULT. The issue-#6 binary instance gives the same three graphs, lambda = 1/10, utility gap, all-profile collisions, rational probability-mixture collisions, and positive-dimensional arrowXYb family. General jointProb_sum proves normalization for every finite categorical model, and a ternary-root worked model exercises a translation and a non-injective local map. MarginClass includes ValidMargin. Equal transforms are machine-checked to give a common zero-regret sign-policy family; the converse reconstruction from an arbitrary optimal-policy oracle remains sourced rather than formalized. HasRegretAtMost is exactly the regret inequality; regret_eq_zero_iff proves support on fibrewise argmax, with ties unconstrained. SCOPE. The public result is RELATED to RE24 Theorems 1-2: this result formalizes no almost-every statement, no gamma(delta), no recovered subgraph, no chart over a mediated CID's parameters, and no Uhler strong-faithfulness theorem. Two of those phrases used to read atlas-wide and no longer can: AISafetyAtlas.Causal.ParameterChart is a real parameter chart, over MAIS's unmediated chance tables rather than a CID's, and AISafetyAtlas.Causal.StructuralModel is a mediated diagram - Everitt AAAI 2021 Definitions 1-5, with decision and utility vertices. Neither is wired to this result, which is why the row still owns the gap. The declaration-by-declaration boundary is recorded in docs/provenance/mais-a2-causal-collision.md.", "original_source_refs": [ "pearl-2009-causality", "richens-everitt-2024", "uhler-etal-2013-faithfulness", "mais-o23-2026", "mais-issue-6-2026" ], "result_shape": "POINT_IMPOSSIBILITY", "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "LocalIntervention", "InterventionProfile", "Mixture", "IsProbabilityMixture", "ProbMixture", "ProbMixture.dirac", "binaryDim", "identityIntervention", "fixIntervention", "fixProfile", "hardInterventionProfile", "flipIntervention", "Model", "Model.factor", "Model.factor_eq_re24", "Model.factor_congr", "Model.factor_fixProfile", "Model.factor_hardInterventionProfile", "Model.jointProb", "Model.jointProb_sum", "Model.jointProb_fixProfile", "Model.jointProb_hardInterventionProfile", "jointProb_sum_two", "Model.Δ", "Model.Δ_fixProfile", "Model.factor_sum", "Model.ParentClosed", "Model.ancestors", "Model.ancestors_eq_univ_iff", "Model.Δmix", "Model.Δmix_eq_sum", "Model.Δmix_congr", "Model.Δmix_eq_on_probMixture_iff", "sum_assignment_two", "Skeleton", "Skeleton.MarginClass", "Model.Δmask", "Skeleton.BehaviorEq", "Skeleton.ValidMargin", "Model.ext", "Skeleton.realizable_iff" ], "relationship": "RELATED", "scope_delta": { "summary": "The collision instantiates a finite categorical CBN over a parametrized value field and an A2 composite; witnesses are computed at Q and transported to R. It does not formalize RE24 Theorems 1-2, an almost-every parameter chart, or Uhler strong faithfulness. Pearl Definition 1.3.1 and a mediated CID both exist elsewhere in the atlas and neither is used here.", "evidence": "docs/provenance/mais-a2-causal-collision.md" }, "module": "AISafetyAtlas.Causal.Model", "build_command": "lake build AISafetyAtlas.Causal.Model AISafetyAtlas.Causal.MarginClass AISafetyAtlas.Examples.Causal.Model AISafetyAtlas.Examples.Causal.BehavioralCollision" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Examples.Causal.margin_class_not_identifiable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Skeleton.MarginClass", "AISafetyAtlas.Causal.Model.Δmix", "AISafetyAtlas.Examples.Causal.collision_mix_edgeless_arrowXY" ], "application": "Machine-checks the pending-review issue-#6 counterexample against the atlas transcription of MAIS-O23: at one skeleton, with nothing observed and margin 1/10, two distinct models of the margin class share a behavioral transform. The upstream problem remains listed as open pending review. As an audit claim this is a possibility result, not a general one: the explicit degeneracy conditions do not by themselves make a causal graph readable off behavior at this skeleton." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.margin_class_not_identifiable_two_graphs", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Examples.Causal.arrowXY_mem", "AISafetyAtlas.Examples.Causal.arrowYX_mem", "AISafetyAtlas.Examples.Causal.collision_mix_arrowXY_arrowYX" ], "application": "The same conclusion under the narrower reading in which the class carries a genuine graph, so the result does not rest on whether the edgeless model is admissible." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.Model.jointProb_sum_shiftCollapse", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Examples.Causal.Model.factor_root", "AISafetyAtlas.Examples.Causal.Model.factor_child", "AISafetyAtlas.Causal.Model.jointProb_sum" ], "application": "Categorical non-vacuity: a ternary root with a binary child is evaluated under a modulo-three translation and a non-injective child map; the numeric preimage factors and general normalization theorem are both exercised." }, { "atlas_declaration": "AISafetyAtlas.Causal.Skeleton.behaviorEq_of_observed_eq_empty", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Skeleton.BehaviorEq", "AISafetyAtlas.Causal.Model.Δmask_empty" ], "application": "What lets a single-transform statement carry the source's family quantifier: with nothing observed, the masked family has one member. The reduction is proved, so the counterexample compares behavioral transforms in the source's sense rather than one component of them." }, { "atlas_declaration": "AISafetyAtlas.Causal.Model.Δmix_congr", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.Δmix_eq_sum", "AISafetyAtlas.Causal.Model.sum_weighted_congr" ], "application": "The step that makes a finite check answer an infinite question: the atlas sigma is a probability mixture of intervention patterns over the model's own value field, so agreement on the sixteen deterministic patterns is only worth stating once it lifts to every mixture in that simplex. Since the decision layer was parametrized over its value field the statement covers real mixtures, which is def:local's simplex of *all* mixtures; no cast is needed and none is claimed." }, { "atlas_declaration": "AISafetyAtlas.Causal.Model.jointProb_hardInterventionProfile", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.factor_hardInterventionProfile", "AISafetyAtlas.Causal.hardInterventionProfile" ], "application": "Hard-intervention fidelity: on any selected set of chance variables, intervened factors become consistency indicators and unselected factors remain, exactly the constructive finite specialization of Pearl's truncated factorization (1.37), over any ordered value field." }, { "atlas_declaration": "AISafetyAtlas.Causal.Model.Δ_fixProfile", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.jointProb_fixProfile", "AISafetyAtlas.Causal.fixProfile" ], "application": "Full hard-intervention corollary: fixing every chance variable produces the Dirac joint at the target assignment, so the utility-gap transform evaluates the gap there." }, { "atlas_declaration": "AISafetyAtlas.Causal.Model.ancestors_eq_univ_iff", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.ancestors", "AISafetyAtlas.Causal.Model.ParentClosed" ], "application": "Faithfulness bridge: certifies that the elimination form of (M5) used throughout is the source's Anc(U) = C, over an ancestor set defined as the least parent-closed superset. The counterexample's own (M5) proof discharges the elimination form directly and does not route through this lemma; what the lemma buys is the right to call that form (M5)." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.transform_identity_edgeless", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.Δ", "AISafetyAtlas.Examples.Causal.g" ], "application": "Non-vacuity: the shared transform takes the value 1/10 under no intervention, so the collision is between behaviors that exist rather than between two identically zero functions." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.Δ_eq_half_sub_joint", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.jointProb_sum_two", "AISafetyAtlas.Examples.Causal.g" ], "application": "Pearl (1.37) plus this g: Delta_M(sigma) = 1/2 - P_M(1,1; sigma). The collision narrative is now a theorem, not prose." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.margin_class_not_identifiable_family", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Skeleton.MarginClass", "AISafetyAtlas.Examples.Causal.collision_edgeless_arrowXYb" ], "application": "The excluded set is not a point. Every value of P(Y=1 | X=0) in [1/10, 7/10] gives a margin-class model colliding with the edgeless one, so the counterexample is a segment. MAIS-O24(c) asks how large the excluded set is; this is a lower bound at one skeleton." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.behaviorEq_has_teeth", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Skeleton.BehaviorEq", "AISafetyAtlas.Examples.Causal.other_mem" ], "application": "Non-vacuity of the collision: two models of the same margin class that behavior DOES tell apart. Without this the agreement theorems could hold because BehaviorEq holds of every pair." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.mm2_shape", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.notMem_parents_self" ], "application": "On two variables a model with any edge carries exactly one of the two arrows -- acyclicity forbidding self-loops and two-cycles. Puts the graph shape in the kernel, which is what lets the two-graph result speak to the agenda's MM_2(lambda) class and hence to MAIS-O34(a)." } ] }, "public": { "group": "What an observer can recover", "summary": "Three different causal pictures of the same two variables can produce exactly the same behavioral transform on every task in a family, including every rational randomized mixture of interventions, while satisfying every explicit non-degeneracy condition the setting imposes. Lean now checks the forward step from equal transforms to one shared zero-regret policy family; the source's converse oracle reconstruction is not claimed. Watching what an agent does can leave its picture of the world undetermined.", "use": "Checking whether a proposed method for reading a world model off behavior has ruled out the cancellations that make behavior blind, or only assumed them away.", "title": "Behavior need not reveal the model", "attribution": "Atlas, over a construction filed against MAIS-O23" } }, { "id": "LAND-CAUSAL-DECISION-001", "name": "Causal decision policies, regret, and identified-set radius", "tags": [ "interpretability", "agent-incentives" ], "result_shape": "INFRASTRUCTURE", "notes": "Generic finite unmediated decision infrastructure. Policy is a probability kernel over the model's value field on any finite decision alphabet, factored through visible chance variables; value is RE24 Section 2.2 expected utility under the Assumption-1 scope fence: decision and utility are not graph vertices here, and DecisionNetwork.expectedUtility_eq_value proves this value equals the printed CID expected utility on every diagram satisfying Assumption 1 (with the Appendix A.2 normalization), which is why the Section 2.2 rows are Same. bestDecision and bestPolicy maximize each visible fibre. regret_decomp and regret_eq_zero_iff prove that zero regret is exactly positive support on fibrewise argmax decisions, leaving ties unconstrained. HasRegretAtMost is the regret inequality used by the A2 query model and the RE24 main text; RE24 Definition 5 instead packages the policy oracle. The Bool specialization derives gap, signPolicy, value_const_sub, and strict-gap action theorems. regret_signPolicy_eq_zero and inIdentifiedSet_zero_of_behaviorEq machine-check the forward A2 bridge from equal transforms to a shared optimal policy family; the converse oracle reconstruction remains outside. InIdentifiedSet, modelError, and IsRadius are A2 query packaging RELATED to, but not a transcription of, RE24 Theorem 2 or gamma(delta). InIdentifiedSet is defined for delta in the value field, and not_inIdentifiedSet_of_neg proves the extension below the source domain delta >= 0 is empty. The value field is a parameter rather than a reduction, so real tables, utilities, policies and mixture weights are in scope; full-assignment indexing and the unmediated projection remain explicit scope reductions. The companion example applies the shared-optimal bridge to the collision, proves the shared-zero-regret relation is not universal, and checks that the empty-visible fibre image is a singleton. The six baseline corpora plus the pinned CausalForge follow-up do not supply this exact combination (NC-009).", "original_source_refs": [ "everitt-etal-2021-agent-incentives", "howard-matheson-2005-influence-diagrams", "richens-everitt-2024" ], "related_result_ids": [ "LAND-CAUSAL-COLLISION-001" ], "relations": [ { "target": "LAND-CAUSAL-COLLISION-001", "kind": "BUILDS_ON", "note": "Consumes the causal Model, derived Skeleton utility gap, margin class, and masked transform layer." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Causal.Policy", "AISafetyAtlas.Causal.Policy.ext", "AISafetyAtlas.Causal.Policy.const", "AISafetyAtlas.Causal.fibreRep", "AISafetyAtlas.Causal.Model.value", "AISafetyAtlas.Causal.Model.value_eq", "AISafetyAtlas.Causal.Model.fibreScore", "AISafetyAtlas.Causal.Model.bestDecision", "AISafetyAtlas.Causal.Model.bestPolicy", "AISafetyAtlas.Causal.Model.optimalValue", "AISafetyAtlas.Causal.Model.value_const_sub", "AISafetyAtlas.Causal.Model.value_le_sign", "AISafetyAtlas.Causal.Model.regret_signPolicy_eq_zero", "AISafetyAtlas.Causal.Model.signPolicy_eq_of_behaviorEq", "AISafetyAtlas.Causal.Model.regret", "AISafetyAtlas.Causal.Model.regret_decomp", "AISafetyAtlas.Causal.Model.regret_eq_zero_iff", "AISafetyAtlas.Causal.Model.HasRegretAtMost", "AISafetyAtlas.Causal.InIdentifiedSet", "AISafetyAtlas.Causal.inIdentifiedSet_zero_of_behaviorEq", "AISafetyAtlas.Causal.inIdentifiedSet_self", "AISafetyAtlas.Causal.inIdentifiedSet_symm", "AISafetyAtlas.Causal.not_inIdentifiedSet_of_neg", "AISafetyAtlas.Causal.modelError", "AISafetyAtlas.Causal.modelError_eq_zero_iff", "AISafetyAtlas.Causal.IsRadius" ], "relationship": "RELATED", "scope_delta": { "summary": "This is the finite unmediated Assumption-1 projection. Decision and utility are not graph vertices here, but DecisionNetwork.expectedUtility_eq_value identifies value with the printed CID expected utility on unmediated diagrams, so the projection is not what keeps this row RELATED. The A2 identified-set/error/radius objects are: they are not RE24 Theorem 2 or its gamma(delta) conclusion. The value field is a parameter, not a restriction: the layer is stated over any ordered field, so it covers def:margin's real tables and def:local's real mixtures directly.", "evidence": "docs/provenance/mais-a2-causal-collision.md" }, "module": "AISafetyAtlas.Causal.Decision", "atlas_module": "AISafetyAtlas.Causal.Decision", "build_command": "lake build AISafetyAtlas.Causal.Decision AISafetyAtlas.Examples.Causal.Decision" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Causal.Model.value_eq", "type": "NEW_PROOF", "source_declarations": [], "application": "Algebraic expansion of the convex combination in the full-assignment representation; not the printed fibre decomposition through Delta^O'." }, { "atlas_declaration": "AISafetyAtlas.Causal.Model.value_const_sub", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Policy.const", "AISafetyAtlas.Causal.Model.Δmix" ], "application": "General binary-decision identity: the advantage of the constant policy D=1 over D=0 is the mixture transform. RE24 Appendix B equation (3), with the opposite sign convention, is one two-variable hard-intervention instance." }, { "atlas_declaration": "AISafetyAtlas.Causal.Model.regret_eq_zero_iff", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.regret_decomp", "AISafetyAtlas.Causal.Model.fibreScore_le_best" ], "application": "The converse formerly embedded in the regret definition: zero regret exactly means that every decision with positive policy mass is fibrewise maximizing; ties remain unconstrained." }, { "atlas_declaration": "AISafetyAtlas.Causal.Model.value_le_sign", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.fibreScore_true_sub_false", "AISafetyAtlas.Causal.Model.signPolicy" ], "application": "Every policy's value is at most the sign-policy value, which is the source max_{pi'} V(pi') direction used to define regret." }, { "atlas_declaration": "AISafetyAtlas.Causal.inIdentifiedSet_zero_of_behaviorEq", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.signPolicy_eq_of_behaviorEq", "AISafetyAtlas.Causal.Model.regret_signPolicy_eq_zero" ], "application": "Machine-checks the forward A2 bridge: equal masked utility-gap transforms give both models the same canonical policy family, with zero regret in each. Reconstructing the numerical transform from an arbitrary optimal-policy oracle is the converse source proposition and is not claimed." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.margin_class_not_identifiable_shared_optimal", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Examples.Causal.margin_class_not_identifiable", "AISafetyAtlas.Causal.inIdentifiedSet_zero_of_behaviorEq" ], "application": "Applies the transform-to-policy bridge to the pending-review MAIS-O23 construction: two distinct margin-class models admit one common zero-regret policy family." }, { "atlas_declaration": "AISafetyAtlas.Causal.not_inIdentifiedSet_of_neg", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.regret_nonneg" ], "application": "Scope control: although the Lean relation accepts every rational tolerance, its extension below RE24 Definition 5's delta >= 0 domain is empty." }, { "atlas_declaration": "AISafetyAtlas.Causal.modelError_eq_zero_iff", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.Model.ext" ], "application": "The graph-or-table error separates models in the chosen finite representation, over any ordered value field." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.not_inIdentifiedSet_high", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.not_inIdentifiedSet_of_opposite_sign", "AISafetyAtlas.Examples.Causal.high_mem" ], "application": "Non-vacuity: the zero-regret identified-set relation does not contain every pair of margin-class models." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.OneNodeClass.modelError_le_ten_mul", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.InIdentifiedSet", "AISafetyAtlas.Causal.modelError" ], "application": "MAIS-O25's linear recovery modulus with the constant exhibited: L = 10 on the one-node margin class. A shared policy family admissible at delta in two models must pay for both on a mixture that centres their advantages, and the two inequalities add with the policy weight cancelling. This is the substantive clause of the antecedent ExactClassAssumptions, which Examples.Conjectures.MAIS assembles into an inhabitant -- so MAIS-O25 is not vacuously true. Nonempty is all that shows: at one vertex only the edgeless graph exists, so several clauses follow from the vertex set rather than from the class." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.card_fibreRep_empty", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.fibreRep" ], "application": "Regression: with nothing visible on two binary variables the fibre image is a singleton, so the decomposition cannot silently count the one fibre four times." } ] }, "public": { "group": "What an observer can recover", "summary": "A policy's utility, regret tolerance, and shared behavior define an identified set of causal models; equal transforms are proved to supply one shared zero-regret policy family, and the residual model error is a radius question.", "use": "Stating interventional value, printed max-regret within the policy type over an ordered value field, of which the printed real case is an instance, and an identified-set relation, without claiming the O27/O28/O34(b) objects.", "title": "Regret and identified-set radius", "attribution": "Atlas infrastructure over MAIS-A2 objects" } }, { "id": "LAND-CAUSAL-STRUCTURAL-001", "name": "Structural causal models, causal influence diagrams, and materiality", "tags": [ "interpretability", "agent-incentives" ], "result_shape": "INFRASTRUCTURE", "notes": "GROUNDING. Everitt, Carey, Langlois, Ortega and Legg, AAAI 2021, Definitions 1 to 5, read from the pinned PDF everitt-etal-2021-aaai-ojs.pdf, sha256 31d4a4b277c562b82000c4d2b81660e89464877ec0cad3549a9375c6d4492598. Definitions 1 and 2 are attributed in print to Pearl 2009 Chapter 7 and are graded against this paper's statement of them, since that is the text transcribed. PEARL 7.1.1, THE DEFINITION PRINT ATTRIBUTES THESE TO, READ 2026-08-22. Pearl's endogenous set is finite - V is a set {V1,...,Vn} with the structural functions indexed i = 1..n - so the Fintype V removed on 2026-08-21 as a narrowing against Everitt was Pearl's own condition, and dropping it widened past both sources rather than repaying a debt. Pearl bounds no domain, so axis B is a narrowing against both and is the one axis both sources agree the atlas invented. And Pearl's clause (iii) ends with the entire set F has a unique solution V(u), footnoted uniqueness is ensured in recursive (i.e., acyclic) systems, Halpern 1998 allows multiple solutions in nonrecursive systems - so unique solvability is part of the object for Pearl and acyclicity is offered as a sufficient condition for it rather than as the content. BONGERS, FORRE, PETERS AND MOOIJ, ANNALS OF STATISTICS 49(5) 2021, 2885-2915, sha256 6ce97700deb27a6e6fc680d0bd8cbfd053f1f67392979afca8e67d9f289a6d31, WAS READ THE SAME DAY as the general treatment of this object - cycles, latent variables, arbitrary domains. Definition 2.1: I is a FINITE index set of endogenous variables and J a disjoint finite index set of exogenous ones, so the most general published version keeps the vertex set finite and the Fintype V removed on 2026-08-21 was a debt to nobody - a free widening with no known consumer rather than a repaid narrowing. Domains are arbitrary standard measurable spaces and f is measurable, which is axis B taken past Fin (dom v) and is where the field actually generalizes; the atlas closed that axis on 2026-09-20 to arbitrary TYPES rather than standard measurable spaces, so its measure layer asks for a measurable-space instance where it needs one, which is less than Bongers and colleagues assume and more than print writes. P_E is a product measure, so axis F is the gap between a product measure and a Finset product rather than between independence and something weaker. And acyclicity is not in the tuple at all - they make no such assumption and allow self-cycles, Definition 2.9 defines acyclic as a property of the graph, Definition 2.3 makes a solution a pair of random variables satisfying the equations almost surely, and Example 2.4 exhibits one SCM with a continuum of solutions and one with none. Unique solvability is a named condition carried where needed. That is the shape axis E produced, arrived at independently. SCM.IsWellFounded is therefore not an atlas addition to the source of record: it is the clause Everitt's paraphrase dropped, carried where it can be discharged instead of assumed everywhere, and chainParents is the proof that the paraphrase loses it at an unbounded vertex set. CONTENT. SCM is structural equations with independent exogenous variables; eval is print's recursion taken literally - well-founded recursion on the parent relation, with eval_eq_f as its fixed-point property, no iteration count and no bound on the number of variables; from 2026-08-22 the well-foundedness is the class SCM.IsWellFounded that eval asks for rather than a field of SCM, and SCM.acyclic carries print's own word. exoJoint_mul_prod is 'mutually independent' in the form an expectation uses. submodel is do(X=x) and softIntervention is the 'more generally' paragraph. CID is Definition 3's graph: a NodeKind partition with childless utility vertices, and IsSingleDecision carried as a hypothesis rather than a field because Definition 3 states no cardinality condition. SCIM is Definition 4, with F a function of a proof that the vertex is not a decision - a total family with an ignored entry at D would give the type more inhabitants than print's tuple has - and with utility domains 'a subset of R' rendered as an injective real readout. Policy is one structural function per decision vertex, and policy_ext_single proves that at Definition 3's single-decision restriction it is print's single pi exactly. withPolicy is M-pi, expectedUtility is E-pi[U] taken in M-pi where print defines it, and optimalValue with exists_isOptimalPolicy is print's max rather than a supremum print never wrote. IsMaterial is Definition 5. The sentence print asserts between Definitions 4 and 5 - for a set of variables X not in Desc_D, Pr-pi(x) is independent of pi and we simply write Pr(x) - is proved rather than assumed: CID.IsDescendant and CID.NotDownstream are the graph condition, SCM.eval_eq_of_f_agree is the content and mentions no policy, SCM.marginal_eq_sum_exo carries a statement about W(e) to one about Pr by summing P(e) over the fibres of eval, and SCIM.marginal_withPolicy_eq_of_notDownstream is print's sentence. It had no row in the coverage audit until 2026-08-22, which was an inventory gap rather than a coverage one: it sits in running prose rather than in a numbered environment. RESULT. Materiality is inhabited, not merely defined: figSCIM is Figure 2a in miniature, where copying the opinion earns 1, every policy of the link-deleted model earns 1/2, and figSCIM_policy_not_const shows the deleted edge is what does the work. Print's Pr(x) is inhabited on the same diagram: figCID_notDownstream_zero puts the opinion outside Desc_D and figSCIM_marginal_opinion_policy_free is the invariance there. SCOPE. Five narrowing axes, and this note claimed two until 2026-08-21 and four until 2026-08-22. The source carries no global finiteness: in the pinned PDF 'finite' occurs exactly twice, both as 'finite-domain', both inside Definition 4, and 'discrete', 'finitely many' and 'countable' do not occur at all. (i) Finitely many variables - CLOSED. SCM, CID and SCIM no longer take Fintype V; the instance moved onto the derived operations that need it, the Finset accessors and the expectation layer, and IsDecision / IsUtility are the unbounded forms with mem_decisions_iff / mem_utilities_iff as the bridge. (ii) Finite domains - CLOSED 2026-09-20. SCM and SCIM carried dom edom : V -> Nat, which on Definitions 1 and 2 had no printed counterpart; from Definition 4 on it is print's own, since 'finite-domain variables' is stated there, and CID carries no domains. The domains are now a family of TYPES. EndoAssignment and ExoAssignment are the assignment spaces at that carrier; AISafetyAtlas.Causal.Assignment keeps the Fin-indexed shape for section 6's decision layer and the two layers do not meet, so no bridge is owed. The positivity field became nonemptiness, since what print's dom(V) being inhabited buys is that the structural functions are well-typed. The marginal-sum field became an UNCONDITIONAL sum, exoProb_tsum, which says what print says at a domain of any size, with exoProb_sum_fintype its finite reading. Every finiteness condition moved onto the derived operation that needs it: the Finset sums of jointProb and marginal, the maximum over policies at Definition 5, and the singleton masses in the measure layer, which additionally needs measurable-space, singleton and countability instances because Fin's were being used implicitly. AdmitsICI deliberately still quantifies over Fin-indexed domain families, so Definition 17 and Theorem 18 say exactly what they said before. WITNESSES: Examples.Causal.StructuralModel.boolFlip has Bool domains rather than an index into Fin 2, and geometricCopy has INFINITE domains with a geometric exogenous law, evaluates by geometricCopy_eval and admits print's Definition 2 by geometricCopy_submodel_eval - where SCM.jointProb cannot be written at all. (iii) Finite indegree - CLOSED 2026-08-22. parents : V -> Set V on both SCM and CID, which is Definition 1's unbounded Pa_V and Definition 3's silence. Nothing in this layer sums over parents, so no finiteness hypothesis was needed; the one visible cost is that a policy's defining property is no longer decidable, so Fintype SCIM.Policy is the classical instance SCIM.instFintypePolicy. (iv) N-ranked acyclicity - CLOSED 2026-08-22, and (iii) and (iv) were one axis rather than two. CID.acyclic is now print's condition verbatim, no vertex reachable from itself along edges, which makes Definition 3 Literal. SCM and SCIM asked instead that the parent relation be well-founded, which is strictly stronger than print's bare word - the axis SHRANK rather than closed at that point, since an N-rank is strictly stronger than well-foundedness once parent sets may be infinite. (v) Well-foundedness as a structure field - CLOSED 2026-08-22, later the same day, and labelled PROVABLY NOT CLOSABLE for part of that day in between. SCM.wellFounded and SCIM.graph_wellFounded are gone; SCM.acyclic is print's word in CID's own Relation.TransGen form; and the requirement lives in the classes SCM.IsWellFounded and CID.IsWellFounded, asked for by eval, eval_eq_f, eval_congr, jointProb and its two lemmas, submodel_eval, submodel_eval_notMem, expectedUtility, IsOptimalPolicy, optimalValue, expectedUtility_le_optimalValue, exists_isOptimalPolicy and IsMaterial, with instIsWellFoundedSubmodel, instIsWellFoundedSoftIntervention, instIsWellFoundedWithPolicy and instIsWellFoundedRemoveInfoLink carrying it across the four constructions. The witness in the tree is unchanged and keeps its job: chainParents n = {n-1} on the integers is acyclic (chainParents_acyclic) and not well-founded (chainParents_not_wellFounded), and on it the equation eval_eq_f asserts has two distinct solutions (chainParents_fixedPoint_not_unique), so Definition 1's the value given by recursive application of the structural functions names nothing unique there. That is why the operations ask for the instance; it is not a reason for the structure to carry a field, and the retracted label confused the two. (vi) The expectation layer is a finite sum - CLOSED 2026-09-20, named 2026-08-22. exoJoint is a product over V, jointProb and expectedUtility are sums over ExoAssignment V edom, and optimalValue is a Finset.sup' of expectedUtility, all of which need Fintype V, while print's P(e) with the exogenous variables mutually independent denotes at unbounded V as a product measure. Definitions 1 and 5 and the policy row carry this; Definition 2 does not, since all four of its declarations omit Fintype V, and Definition 4 does not, since its column is the structure alone. An intermediate draft graded these rows Same with Bridged fidelity; the audit table has no fidelity column and a field print does not write makes a row Narrower, so that grade was withdrawn the same day, and what closed Definition 4 in the end was moving the field rather than arguing about it. wellFounded_iff_exists_rank is the witness that axes (iii) and (iv) had to move together: while parents was a Finset, well-foundedness and an N-valued rank were equivalent, so dropping the rank alone would have admitted precisely the same models. Definitions 3 and 4 are Same as of 2026-08-22 and are the two of the five that closed - Definition 4 last, when the well-foundedness field came off SCIM, and alone among the survivors because its atlas column is Causal.SCIM and nothing else. Definitions 1 and 2 remain Narrower on domains alone; Definition 5 and the policy row closed on 2026-09-20 when the expectation layer did. The remaining axis is in the public types, so the row is graded on the artifact - which is also why the audit grades a row on every declaration in its atlas column rather than on the printed object's nearest counterpart alone. A CLAIM THIS NOTE USED TO MAKE, RETRACTED. It said axis (i) was an implementation choice because well-founded recursion on the acyclicity rank would evaluate an infinite diagram of finite-rank nodes. The conclusion was right and the reason was wrong about the code: eval was evalIter stopped after Fintype.card V applications, so rank recursion was a rewrite and not a relabelling. Axis (i) closed without that rewrite, because the structures' own fields never needed the instance. The rewrite itself was done on 2026-08-22 for axes (iii) and (iv), and eval no longer takes [Fintype V] at all; evalIter and the two rank lemmas that supported it are deleted. ONE DECLARATION DEVIATES IN BOTH DIRECTIONS. SCIM.Policy is Mixed rather than Narrower: narrower because expectedUtility sums over ExoAssignment V edom, which wants Fintype V - the axes (iii), (iv) and (v) half of this narrowing closed on 2026-08-22, withPolicy now hands SCM print's own CID.acyclic and instIsWellFoundedWithPolicy supplies the recursion's hypothesis; and exists_isOptimalPolicy needing the policy type finite was withdrawn from this list the same day, because print writes a maximum and states no condition delivering one, so finiteness supplies print's assertion rather than cutting below it; and wider because print's policy paragraph is a single pi under D = {D} while the atlas defines a family over the decision subtype, so it reaches multi-decision diagrams where print defines nothing. policy_ext_single keeps that widening disclosed. NOT HERE, AS OF 2026-09-10, AND THIS LIST WAS WRONG UNTIL THEN. It read 'd-separation (Definition 6), and therefore Theorems 9, 14, 16 and 18 ... Desc_V and the directed-path relation', and four of those entries had been false since 2026-09-09. Definition 6 is CID.DSepPath and CID.DSepSet in AISafetyAtlas.Causal.DSep, graded Same in coverage-audit section 8 as of 2026-09-20. It was graded Narrower between 2026-09-10 and 2026-09-20, because both new modules carried Fintype V and Finset endpoint sets where print bounds no vertex set, which reopened the axis (i) this row records as CLOSED. THAT REOPENING CLOSED AGAIN ON 2026-09-20: CID.DSepSet states Definition 6 on Set V at an unbounded carrier, with CID.UAdj, CID.IsCollider, CID.IsChainOrFork, CID.IsWalk and CID.Blocked as print's own clauses, and CID.dSepSet_iff_dSep proves the Finset predicate CID.DSep is what it comes to on a Fintype - so the Bayes-Ball form is the computation of Definition 6 rather than a second reading of it, exactly as mem_decisions_iff is for CID.decisions, and axis (i) is CLOSED across this section again. Examples.Causal.DSep.intChain is the integer chain and Examples.Causal.DSep.intChain_not_dSepSet answers Definition 6 on a carrier no Bayes-Ball statement can be asked about. Two findings came with it: clause 1's 'W is not in Z' is redundant against 'no descendants of W are in Z' because CID.IsDescendant is reflexive (CID.blocked_collider_iff), and clause 2 is clause 1's complement ONLY under acyclicity, since a chain X to W to Y is also a collider when Y points to W, which is a two-cycle (CID.isChainOrFork_iff_not_isCollider). THE WALK-VERSUS-PATH AXIS CLOSED THE SAME DAY, which takes Definition 6 to Same and Definition 7 to Wider and leaves neither row owing a narrowing state. CID.IsWalk asks for no Nodup, so the statements quantify over walks where print says path, which makes the predicate hold of fewer triples and is therefore a narrowing; CID.DSepPath is print's word and CID.dSepSet_iff_dSepPath proves the two readings are the same predicate. The proof is the splicing argument: delete the segment between two occurrences of a repeated vertex, and exactly one triple of the result is new. The hard case is where the repeated vertex is unconditioned, unactivated and a collider on the spliced path - then neither old triple at it can be a collider, so it points forwards on both sides, and forward_of_not_activated follows the deleted segment to its far end, whose own triple is then a collider the original path's activity forbids. CID.Activated is the predicate that argument pushes along an edge and CID.not_blocked_iff_isActive separates clause 3 from clauses 1 and 2 so the splice can preserve them independently. Upstream supplied nothing here either: Nodup does not occur under Causalean/Graph/, so the standard fact Bayes-Ball correctness rests on was not importable. Definition 7 is CID.IsNonrequisiteSet in AISafetyAtlas.Causal.Requisite, graded Wider since 2026-09-20, both of its inherited axes closed that day through CID.utilityDescendantsSet, CID.requisiteContextSet, CID.isNonrequisiteSet_iff and CID.isNonrequisiteSet_iff_dSepPath; Definition 17 is SCIM.HasICIAt and Theorem 18 is CID.admitsICI_iff, both in AISafetyAtlas.Causal.Incentive and both graded Mixed, with Theorem 18 closed in both directions. Theorem 18 needed no d-separation at all - its path runs from the decision through X in G rather than in G_min - which is why it closed first. The descendant relation is CID.IsDescendant and the path-through relation is SCIM.PathThrough. Those four modules are outside this row's atlas column, which is Causal.StructuralModel alone, and they hold no registry row of their own; that absence is the reason a stale NOT HERE survived every green gate for a day, and it is recorded as debt in docs/provenance/source-coverage-audit.md. WHAT IS GENUINELY NOT HERE: Theorem 9 (value of information), Definition 11 (minimal reduction), Theorem 12 (response incentive), Definition 13 (counterfactual fairness), Theorem 14 and Theorem 16 (value of control) - the three criteria that need the minimal reduction, hence Definition 11, hence which edges Definition 7 deletes, plus their Lemmas 23, 25, 26, 27 and 28. Nested counterfactuals and multi-decision imputation. Nothing sends a SCIM's decision vertex to Causal.Model.value, so LAND-CAUSAL-DECISION-001 keeps its own unmediated grade and Richens and Everitt Section 2.2 is still that projection. PROBABILITY LAWS WITHOUT A FINITE VERTEX SET, 2026-09-14. SCM.exoLaw gives the independent exogenous draws as Mathlib's infinite product measure over an ARBITRARY vertex type, which is a probability measure even where the assignment space is uncountable; exoLaw_map_eval recovers each original marginal and SCM.exoPMF is the existing exoProb as a PMF. THE FINITE CASE IS RECOVERED, NOT ABANDONED: exoLaw_eq_pi identifies it with Mathlib's finite product and exoLaw_singleton with the model's own exoJoint weights, so every Finset-level statement in this row is an instance of the general one rather than a competitor to it. SCIM.observableLaw_withPolicy_eq_of_notDownstream is print's policy-invariance sentence at that level, at an arbitrary Set of observables, and SCIM.exoLaw_withPolicy_eq is the exogenous half by rfl. THE PRICE IS NAMED: SCM.observableLaw takes a measurability hypothesis, because an arbitrary function of infinitely many parents need not be measurable, and a nonmeasurable map must not silently return zero. At print's own finite setting that hypothesis is free - SCM.measurable_exo discharges it, since the exogenous space is then a finite product of finite discrete spaces - so nothing print asserts is bought with a condition print lacks. WITNESSES: Examples.Causal.StructuralModel.figSCIM_observableLaw_opinion_policy_free instantiates the general theorem on print's own Figure 2a with both side conditions discharged, and Examples.Causal.StructuralModel.independentBits_first_zero inhabits the product layer OUTSIDE the finite class, at countably many independent bits, computing a cylinder probability. Section 8's policy-invariance row went Mixed to Wider on this. THE EXPECTATION LAYER READ THAT LAW ON 2026-09-20, and until then it had not: exoLaw existed and expectedUtility went on summing. SCIM.expectedUtilityLaw is E-pi[U] as an integral against exoLaw at an unbounded vertex set, SCM.endoLaw is print's 'this induces a joint distribution' as a distribution, and SCIM.expectedUtilityLaw_eq and SCM.endoLaw_singleton prove the finite sums COMPUTE them - the same shape as CID.DSep computing CID.DSepPath and mem_decisions_iff recovering CID.decisions. TWO HYPOTHESES AND NEITHER BOUNDS V: the utility vertices are a Fintype on the subtype, which is print's own sum over U needing to denote rather than a cardinality condition on V; and measurability is asked for, since a Bochner integral of a nonmeasurable function is silently zero, discharged free at finite V by measurable_exo. ATTAINMENT IS SPLIT RATHER THAN CLAIMED: SCIM.IsOptimalPolicyLaw is print's 'any policy that maximises E-pi[U]' at an unbounded vertex set, because that is a comparison; exists_isOptimalPolicy is 'a maximiser exists' and keeps the finite policy space print states no condition for, and optimalValue_eq_sup'_law shows the maximand is print's expectation. WITNESSES: Examples.Causal.StructuralModel.independentBits_endoLaw_first_zero answers print's induced joint at V = Nat, where jointProb cannot be written at all, and figSCIM_optimalValue_law and figSCIM_isOptimalPolicyLaw_copy restate the computed values under print's own object. STILL NOT HERE: Definition 17's conditioning. SCIM.condExp is a ratio of sums over SCIM.contextFiber, a filtered Finset.univ, and generalising it is NOT the same axis - print's E[. | pa^D] conditions on the event Pa^D = pa^D, which once Pa^D may be infinite can be null under exoLaw, where print's conditional expectation names nothing; print writes Pr(pa^D) > 0 at Definition 13 and writes no condition at Definition 17. So it needs a conditional expectation given a sigma-algebra plus a reading of Definition 17 that print does not supply. Summable utility families over an unbounded U are likewise absent and are job 2 rather than this axis.", "original_source_refs": [ "everitt-etal-2021-agent-incentives" ], "related_result_ids": [ "LAND-CAUSAL-DECISION-001", "LAND-CAUSAL-COLLISION-001" ], "relations": [ { "target": "LAND-CAUSAL-DECISION-001", "kind": "BOUNDARY_PARTNER", "note": "Parallel, not layered. Causal.Decision is the unmediated Assumption-1 projection and this is the mediated diagram; no declaration connects them. That missing map no longer holds Richens and Everitt Section 2.2 back: those rows closed on 2026-09-20 through Causal.DecisionNetwork instead (expectedUtility_eq_value, regret_eq_value_regret), and are graded Same in section 6 of the coverage audit." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Causal.SCM", "AISafetyAtlas.Causal.wellFounded_of_rank", "AISafetyAtlas.Causal.acyclic_of_rank", "AISafetyAtlas.Causal.wellFounded_iff_exists_rank", "AISafetyAtlas.Causal.chainParents", "AISafetyAtlas.Causal.chainParents_acyclic", "AISafetyAtlas.Causal.chainParents_not_wellFounded", "AISafetyAtlas.Causal.chainParents_fixedPoint_not_unique", "AISafetyAtlas.Causal.SCM.exoJoint", "AISafetyAtlas.Causal.SCM.exoJoint_sum", "AISafetyAtlas.Causal.SCM.exoJoint_mul_prod", "AISafetyAtlas.Causal.SCM.eval", "AISafetyAtlas.Causal.SCM.eval_eq_f", "AISafetyAtlas.Causal.SCM.eval_congr", "AISafetyAtlas.Causal.SCM.jointProb", "AISafetyAtlas.Causal.SCM.jointProb_sum", "AISafetyAtlas.Causal.SCM.submodel", "AISafetyAtlas.Causal.SCM.submodel_eval", "AISafetyAtlas.Causal.SCM.submodel_eval_notMem", "AISafetyAtlas.Causal.SCM.softIntervention", "AISafetyAtlas.Causal.NodeKind", "AISafetyAtlas.Causal.CID", "AISafetyAtlas.Causal.CID.decisions", "AISafetyAtlas.Causal.CID.utilities", "AISafetyAtlas.Causal.CID.structureNodes", "AISafetyAtlas.Causal.CID.observations", "AISafetyAtlas.Causal.CID.decisions_disjoint_utilities", "AISafetyAtlas.Causal.CID.IsSingleDecision", "AISafetyAtlas.Causal.SCIM", "AISafetyAtlas.Causal.SCIM.Policy", "AISafetyAtlas.Causal.SCIM.policy_ext_single", "AISafetyAtlas.Causal.SCIM.withPolicy", "AISafetyAtlas.Causal.SCIM.withPolicy_f_mem", "AISafetyAtlas.Causal.SCIM.withPolicy_f_notMem", "AISafetyAtlas.Causal.SCIM.expectedUtility", "AISafetyAtlas.Causal.SCIM.IsOptimalPolicy", "AISafetyAtlas.Causal.SCIM.optimalValue", "AISafetyAtlas.Causal.SCIM.expectedUtility_le_optimalValue", "AISafetyAtlas.Causal.SCIM.exists_isOptimalPolicy", "AISafetyAtlas.Causal.SCIM.removeInfoLink", "AISafetyAtlas.Causal.SCIM.removeInfoLink_sub", "AISafetyAtlas.Causal.SCIM.instFintypePolicy", "AISafetyAtlas.Causal.SCIM.IsMaterial", "AISafetyAtlas.Examples.Causal.StructuralModel.figSCIM", "AISafetyAtlas.Examples.Causal.StructuralModel.figSCIM_optimalValue", "AISafetyAtlas.Examples.Causal.StructuralModel.figCut_optimalValue", "AISafetyAtlas.Examples.Causal.StructuralModel.figSCIM_opinion_isMaterial", "AISafetyAtlas.Examples.Causal.StructuralModel.figSCIM_policy_not_const", "AISafetyAtlas.Examples.Causal.StructuralModel.copyChain_eval_one", "AISafetyAtlas.Examples.Causal.StructuralModel.utility_childless_has_teeth" ], "relationship": "RELATED", "scope_delta": { "summary": "Definitions 1 to 5 at finite domains on Definitions 1 and 2, which print does not require until Definition 4 and then only of the domains, and at finite indegree and an N-valued acyclicity rank throughout, neither of which print imposes anywhere. The finite-variable-set axis closed on 2026-08-21: SCM, CID and SCIM carry no Fintype V, and the instance sits on the derived operations that need it. CID therefore narrows on the two graph axes and on nothing else, since it carries no domains. SCIM.Policy is the one declaration that deviates in both directions - narrower on the carrier, wider because it is a family over the decision subtype where print writes a single policy under a single-decision restriction. This summary grades Definitions 1 to 5 on Causal.StructuralModel and nothing else. It said 'No incentive concept and no graphical criterion is formalized' until 2026-09-10; that was false from 2026-09-09, when Causal.DSep, Causal.Requisite and Causal.Incentive landed with Definition 6, Definition 7, Definition 17 and Theorem 18. Those modules are graded in coverage-audit section 8 and carry no registry row of their own.", "evidence": "docs/provenance/source-coverage-audit.md" }, "module": "AISafetyAtlas.Causal.StructuralModel", "atlas_module": "AISafetyAtlas.Causal.StructuralModel", "build_command": "lake build AISafetyAtlas.Causal.StructuralModel AISafetyAtlas.Examples.Causal.StructuralModel" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Causal.SCM.eval_eq_f", "type": "NEW_PROOF", "source_declarations": [], "application": "Print says W(eps) is given by 'recursive application of the structural functions'. The atlas iterates F once per variable; this proves the result is a fixed point of F, which is what makes the unrolling print's recursion rather than an arbitrary finite one." }, { "atlas_declaration": "AISafetyAtlas.Causal.SCM.exoJoint_mul_prod", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.SCM.exoJoint" ], "application": "Independence in the form an expectation uses: the joint expectation of a product of per-variable factors is the product of the marginal expectations. exoJoint_sum is the constant case. Print states independence and never uses it in this form." }, { "atlas_declaration": "AISafetyAtlas.Causal.SCIM.policy_ext_single", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.SCIM.Policy", "AISafetyAtlas.Causal.CID.IsSingleDecision" ], "application": "Print writes one policy because it has restricted to D = {D}. The atlas carries a family indexed by decision vertices, to keep withPolicy free of transports between Fin (dom v) and Fin (dom d); this proves the family is print's single datum at print's own restriction." }, { "atlas_declaration": "AISafetyAtlas.Causal.SCIM.exists_isOptimalPolicy", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.SCIM.optimalValue" ], "application": "Definition 5 writes V*(M) as a maximum over policies. Without an attainment theorem the atlas value is a supremum print never wrote; this exhibits a policy reaching it, and it exists only because the policy type is finite." }, { "atlas_declaration": "AISafetyAtlas.Causal.SCIM.observableLaw_withPolicy_eq_of_notDownstream", "type": "NEW_PROOF", "source_declarations": [ "exoLaw", "observableLaw", "measurable_exo", "observableLaw_withPolicy_eq_of_notDownstream" ], "application": "Agent incentives: print's policy-invariance sentence at the level of probability laws, with no finite vertex set and an arbitrary set of observables. The exogenous draws are a product measure, the finite joint weights are recovered by exoLaw_eq_pi and exoLaw_singleton, and the measurability side condition is discharged for free at print's own finite setting by measurable_exo." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.StructuralModel.figSCIM_opinion_isMaterial", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.SCIM.IsMaterial", "AISafetyAtlas.Causal.SCIM.removeInfoLink" ], "application": "Definition 5 is a condition an observation meets: print's Figure 2a in miniature, where deleting the information link costs the agent half its utility. Without it IsMaterial could be a definition nothing satisfies." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.StructuralModel.figSCIM_policy_not_const", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Examples.Causal.StructuralModel.figCut_optimalValue" ], "application": "The deleted edge doing work. Reading the opinion is possible in M and impossible in M with the link removed; without this the materiality witness is 1/2 < 1 between two numbers on two structures whose difference was never exercised." }, { "atlas_declaration": "AISafetyAtlas.Examples.Causal.StructuralModel.utility_childless_has_teeth", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Causal.CID" ], "application": "Definition 3's 'utility nodes have no children' is a field rather than a consequence; this exhibits the same graph failing that clause and no other." } ] }, "public": { "group": "What an observer can recover", "summary": "A causal influence diagram with decision and utility vertices, structural equations with exogenous noise, policies as structural functions, and the observations whose removal costs the agent utility.", "use": "Stating Everitt et al.'s setup - submodels, policies, expected utility in the induced structural model, and materiality - without claiming any incentive concept or graphical criterion.", "title": "Causal influence diagrams and material observations", "attribution": "Atlas transcription of Everitt et al., AAAI 2021, Definitions 1-5" } }, { "id": "LAND-SOV-TRULYPLAYABLE-001", "name": "Truly playable effectivity functions, and the finite-domain corollary", "tags": [ "social-choice", "multi-agent" ], "notes": "GROUNDING. Goranko, Jamroga and Turrini, JAAMAS 26: 288-314, 2013, Definitions 4, 5, 8 and 9 and Propositions 1, 2, 5 and 6, read from rendered pages of the pinned author manuscript because text extraction drops mathematics from this file. Graded statement by statement in section 14 of docs/provenance/source-coverage-audit.md, which is the first section this source has had. CONTENT. nonmonotonicCore is Definition 4, the minimal sets of E(C), which print attributes to Pauly. CompleteCore is Definition 5 at an arbitrary coalition. TrulyPlayable is Definition 8 - playable, and complete core at the EMPTY coalition only, which is print's own binder and not a simplification: print's Example 5 exhibits a truly playable function whose core at a singleton coalition is empty. IsCrown is Definition 9. Playable.emptyFilter is Proposition 1 (1) as the object print names rather than as four closure lemmas - print's footnote 2 lists exactly Mathlib's Filter together with properness, and Playable.emptyFilter_neBot is that properness. Playable.subsingleton_nonmonotonicCore_empty is Proposition 1 (2), and its load-bearing step is that a minimal element of E(empty) is a least one. nonmonotonicCore_effectivity_empty and trulyPlayable_effectivity are Proposition 2. Proposition 5 (1) iff (2) is trulyPlayable_iff_nonmonotonicCore_empty_nonempty; (1) iff (3) is stated twice, as trulyPlayable_iff_principal by exhibiting a least element and as trulyPlayable_iff_emptyFilter_principal through Mathlib's Filter.principal. RESULT. Proposition 6 - on finite domains playability and true playability coincide - is Playable.trulyPlayable_of_finite, and since 2026-09-10 it is proved the way print proves it, in print's own one line: by Proposition 5.3 and the fact that every filter on a finite set is principal, the second half being Mathlib's. SCOPE. Wider than print at exactly one place. The argument consumes strictly less than a finite domain: Playable.trulyPlayable_of_exists_minimal asks only that the family E(empty) have a minimal element and uses no finiteness anywhere, trulyPlayable_of_finite_empty asks only that the family be finite, and print's hypothesis is two corollaries down. cofiniteEff witnesses that the hypothesis cannot be dropped altogether. A CLAIM RETRACTED 2026-09-10. The module said print gives no route from finiteness to true playability and attributed that route to the unlicensed Lean 3 development kaiobendrauf/cl-lean. It is Proposition 6, on the page the module header says it read. Nothing in the tree checked that attribution because this source had no coverage section and this module had no registry row; both now exist, and that is what the retraction cost. NOT HERE. Proposition 5 (4), the crown equivalence - IsCrown is defined and the equivalence is unproved. Section 4.2, the corrected representation theorem, which is the headline result of the repair: paulyGame in PlayableConverse is Pauly's construction unrevised, so the atlas is one step short of it. Section 4.4, the reconstruction. Example 5.", "original_source_refs": [ "atlas-ref-gjt-2013-truly-playable" ], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "nonmonotonicCore", "CompleteCore", "TrulyPlayable", "IsCrown", "Playable.emptyFilter", "Playable.emptyFilter_neBot", "Playable.subsingleton_nonmonotonicCore_empty", "nonmonotonicCore_effectivity_empty", "trulyPlayable_effectivity", "Playable.trulyPlayable_iff_nonmonotonicCore_empty_nonempty", "Playable.trulyPlayable_iff_principal", "Playable.trulyPlayable_iff_emptyFilter_principal", "Playable.trulyPlayable_of_finite", "Playable.trulyPlayable_of_finite_empty", "Playable.trulyPlayable_of_exists_minimal", "nonmonotonicCore_cofiniteEff_empty", "not_trulyPlayable_cofiniteEff" ], "module": "AISafetyAtlas.Sovereignty.TrulyPlayable", "build_command": "lake build AISafetyAtlas.Sovereignty.TrulyPlayable", "relationship": "RELATED", "scope_delta": { "summary": "Definitions 4, 5, 8 and 9 and Propositions 1, 2, 5 (1) iff (2) and (1) iff (3), and 6, at print's own binders. Wider at one place: Proposition 6 is proved from a minimal element of E(empty) rather than from a finite domain, so trulyPlayable_of_exists_minimal uses no finiteness at all and print's hypothesis is two corollaries down. RELATED rather than EXACT because the paper's constructive half is absent: Proposition 5 (4), the corrected representation theorem of section 4.2, the reconstruction of section 4.4 and Example 5 are all No rows in section 14 of the coverage audit.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.Playable.trulyPlayable_of_finite", "type": "NEW_PROOF", "source_declarations": [ "Playable.trulyPlayable_of_finite", "Playable.trulyPlayable_of_exists_minimal", "not_trulyPlayable_cofiniteEff" ], "application": "Cognitive sovereignty: which effectivity functions are the powers of an actual game, and therefore which claims about what a coalition can force are realizable. Goranko, Jamroga and Turrini (JAAMAS 2013) Proposition 6." } ] }, "public": { "group": "Aggregation and multi-agent structure", "summary": "Not every consistent description of who can force what is the description of an actual game.", "use": "When checking whether a claimed distribution of power could be realized at all.", "title": "Truly playable effectivity functions", "attribution": "Goranko, Jamroga & Turrini" } }, { "id": "LAND-SOV-PLAYABILITY-001", "name": "Pauly's playability conditions, the easy direction, and the converse that fails", "tags": [ "social-choice", "multi-agent" ], "notes": "GROUNDING. Pauly, J. Logic and Computation 12(1): 149-166, 2002, section 3, read from rendered pages 152 and 155 because text extraction strips every formula from this file. Graded in section 15 of docs/provenance/source-coverage-audit.md, the first section this source has had. Print is explicit that it does NOT assume the outcome function surjective and does NOT assume X in E(N) for all nonempty X, unlike the two earlier characterizations it cites, so the atlas inherits that generality from print rather than adding it. CONTENT. Playable is print's five conditions as five fields in print's order. Playable.regular and Playable.mono_coalition are Lemma 3.1. effectivity_playable and playable_effectivity are the easy direction of Theorem 3.2, game to playable, with forces_univ_of_not_forces_empty_compl carrying N-maximality, the one condition that is not immediate. Individualistic is section 3.2's definition, and iUnion_effectivity_singleton_subset records that one inclusion of print's equality is free. BlockChoice and the refinement sequence through exists_refineFixed are Pauly's page-153 construction, built and proved part by part. RESULT, AND IT IS NEGATIVE. The converse half of Theorem 3.2 - every playable effectivity function is some strategic game's - is FALSE at an infinite state set. not_exists_gameForm_cofiniteEff proves it false, on Goranko, Jamroga and Turrini's cofinite effectivity function over the naturals. The page-153 construction is therefore machinery whose target statement does not hold; it is kept because section 4.2 of the repair paper revises exactly it. Theorem 3.3 inherits the defect: individualistic_iff_exists_dictator is the biconditional on an EXISTING game form, where print quantifies over an arbitrary effectivity function, and print's own proof of the surviving direction invokes Theorem 3.2. The docstring stated print's quantifier until 2026-09-10 and now states what the theorem is. NOT HERE. Lemma 3.4, playable iff semi-playable, regular and N-maximal - and the reason recorded for not building it was wrong. cognitive-sovereignty-obligation.md said the provenance of semi-playability was unclear because an unlicensed Lean 3 development cites three papers without saying which owns it. It is Pauly's own, section 3.3, page 155, in the source this row grades. Nothing checked that because Pauly had no coverage section until today. Also not here: the independence of the five conditions, which print asserts and does not prove, and the whole of section 4, the modal layer. CONDITION (ii) ADDED 2026-09-20. effectivity_univ_eq_nonempty is Peleg's remaining EF condition, E(N) = 2^A minus the empty set, under the same surjectivity hypothesis effectivity_empty_eq_singleton_univ carries. It is a theorem here rather than a stipulation: the grand coalition fixes a whole profile, so it forces exactly the sets containing a reachable outcome, and surjectivity makes that every non-empty set. Witnessed at AISafetyAtlas.Examples.Sovereignty.effectivity_shirtGame_univ, where the surjectivity comes from the representation itself rather than from an assumption.", "original_source_refs": [ "atlas-ref-pauly-2002-coalition-logic" ], "related_result_ids": [ "LAND-SOV-TRULYPLAYABLE-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "effectivity_playable", "not_forces_empty", "forces_univ_of_not_forces_empty_compl", "not_forces_compl_of_forces", "Individualistic", "iUnion_effectivity_singleton_subset", "individualistic_iff_exists_dictator", "BlockChoice", "forces_iInter_blockChoice", "iInter_blockChoice_nonempty", "exists_refineFixed", "effectivity_univ_eq_nonempty" ], "module": "AISafetyAtlas.Sovereignty.Playability", "build_command": "lake build AISafetyAtlas.Sovereignty.Playability", "relationship": "RELATED", "scope_delta": { "summary": "The five playability conditions, Lemma 3.1 and the easy direction of Theorem 3.2 at print's own binders, including print's deliberate refusal of surjectivity. Theorem 3.3 is stated here on an existing game form where print quantifies over an arbitrary effectivity function, and THAT GAP IS CLOSED BY COUNTEREXAMPLE AS OF 2026-09-20 rather than owed: print's Theorem 3.3 is FALSE at print's own quantifier, refuted in AISafetyAtlas.Sovereignty.PlayableConverse by not_forall_individualisticEff_exists_dictator at the same cofinite witness that refutes Theorem 3.2's converse. What stands here is print's statement restricted to effectivity functions that are some game form's, which is the restriction print's own proof performs in its first sentence. RELATED rather than EXACT because the converse of Theorem 3.2 is refuted rather than proved, Lemma 3.4 is absent, and the independence of the five conditions is unproved.", "evidence": "docs/provenance/source-coverage-audit.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Playable", "playable_effectivity", "Playable.regular", "Playable.mono_coalition", "cofiniteEff", "cofiniteEff_playable", "not_exists_gameForm_cofiniteEff", "IndividualisticEff", "individualisticEff_effectivity_iff", "cofiniteEff_individualisticEff", "not_forall_individualisticEff_exists_gameForm", "not_forall_individualisticEff_exists_dictator" ], "module": "AISafetyAtlas.Sovereignty.PlayableConverse", "build_command": "lake build AISafetyAtlas.Sovereignty.PlayableConverse", "relationship": "RELATED", "scope_delta": { "summary": "Print's playability predicate and Lemma 3.1, plus the REFUTATION of the converse half of Theorem 3.2 at an infinite state set. The refutation is Goranko, Jamroga and Turrini's and is graded against them in section 14; this module carries their counterexample and the conclusion. Nothing here covers the printed converse, and the No row for it in section 15 is a false theorem rather than outstanding work. THEOREM 3.3 IS REFUTED HERE TOO, ADDED 2026-09-20: IndividualisticEff is print's own definition of individualistic on a bare effectivity function - print's playability conjunct included, which Playability.Individualistic drops because it is free for a game form - and individualisticEff_effectivity_iff is the lemma that the two agree on a game form. cofiniteEff_individualisticEff says the same witness satisfies it, for the cheapest reason: on one player the union over individuals IS the grand coalition. So not_forall_individualisticEff_exists_dictator refutes print's Theorem 3.3 at print's own quantifier, and not_forall_individualisticEff_exists_gameForm does it without reading the word dictatorship at all, since the witness is outside the range of effectivity altogether. Section 15's Theorem 3.3 row was NARROWER and is now closed by counterexample.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.not_exists_gameForm_cofiniteEff", "type": "NEW_PROOF", "source_declarations": [ "not_exists_gameForm_cofiniteEff", "effectivity_playable", "individualistic_iff_exists_dictator" ], "application": "Cognitive sovereignty: which descriptions of who can force what are the description of an actual game. Pauly (J. Logic and Computation 2002) Theorem 3.2, whose converse fails at an infinite outcome set." } ] }, "public": { "group": "Aggregation and multi-agent structure", "summary": "Pauly's five conditions describe the power of a game, but they do not pin one down.", "use": "When a claimed distribution of power has to be checked against an actual game.", "title": "Playability and its failed converse", "attribution": "Pauly; refutation by Goranko, Jamroga & Turrini" } }, { "id": "LAND-SOV-RETARGETABLE-001", "name": "Retargetable decision-makers have orbit-level tendencies, up to theorem A.13", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. Turner and Tadepalli, arXiv:2206.13477v2, section 3 and appendix B.2 with definitions A.6, A.7, A.12 and theorem A.13, read from rendered pages 4, 5, 12, 14-21. Graded in section 16 of docs/provenance/source-coverage-audit.md, the first section this source has had. CONTENT. Section 3 is finite combinatorics on group orbits: MostOrbit is definition 3.2, MultiplyRetargetable is definition 3.5's three clauses in print's order, SimplyRetargetable is definition 3.3 at print's quantifier order - ONE permutation for every parameter, not one per parameter - and MultiplyRetargetable.mostOrbit is theorem 3.6 by print's disjoint-image counting argument. SimplyRetargetable.mostOrbit is proposition 3.4; print's footnote 3 says definition 3.3 implicitly assumes Theta closed under permutation, and the Lean carries that as the explicit hypothesis hclosed, so the total assumptions match print. The appendix chain is mostOrbit_of_orbitConditions (B.7), SupersetCopies (B.8), mostOrbit_of_supersetCopies (B.9), increasingUnderJointPerm_of_hide and its invariant half (B.10), EUDetermined (A.12), EUDetermined.jointPermInvariant (B.11) and eu_determined_mostOrbit (A.13). WIDER, AND ONE OF THEM CORRECTS PRINT. The group is arbitrary where print's is S_d; instantiate at Equiv.Perm (Fin d) to recover print. B.11 is proved against an arbitrary invariant pairing, because print's proof of it uses exactly one property of the permutation matrix - orthogonality, at equation (47). A.13 asks B.8's superset-copies rather than A.7's copies, which is strictly looser, and InvolutiveCopies.supersetCopies bridges print's hypothesis to it. AND: lemma B.7's item 1 is stated WITHOUT print's guard, because print's own proof requires the unguarded form - equation (24) rewrites through the involution and equation (25) applies item 1 at phi_i-inverse dot theta-star, where the guard would be the negation of what equation (29) concludes. Lemma B.9, print's only consumer, discharges item 1 with no antecedent. Print's table 3 shows item 4's guard is NOT droppable, and it is kept. NARROWER, COSTED. Definition A.12's carrier is print's power set of R^d and the Lean's is Finset V. The cost buys nothing: no step of the chain reads a cardinality, an infinite profile would need a multiplicity function into Cardinal that Mathlib does not carry, and print's own equation (4) presupposes finiteness by writing a multiset and its size. NOT HERE, AND THE GAP IS ONE LEMMA WIDE. Proposition A.11, the paper's headline result over seven rationalities, has no target; each item is an instance of A.13 once the rationality is rewritten as an EU-determined function. The single missing step to print's distribution-level results is lemma B.5, which lifts a joint-permutation-increasing function through an expectation. Also not here: lemmas B.1, B.2, B.3, B.6, definition B.12, appendix C, and the whole power layer at definitions D.7 to D.10 and theorem D.11. THIS ROW IS NOT EVIDENCE OF A POWER-SEEKING TENDENCY: print's footnote 2 says A and B are labels demanding no structure, and the power reading enters only at appendix D, which rests on Turner et al. 2021.", "original_source_refs": [ "atlas-ref-turner-tadepalli-2022-retargetable" ], "related_result_ids": [ "LAND-DEC-MDP-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "orbitIn", "orbitFavouring", "orbitAgainst", "MostOrbit", "MultiplyRetargetable", "SimplyRetargetable", "orbitIn_finite", "MultiplyRetargetable.mostOrbit", "SimplyRetargetable.mostOrbit", "SimilarUnder", "ContainsCopies", "containsCopies_one_of_subset", "ContainsCopies.mono" ], "module": "AISafetyAtlas.Sovereignty.Retargetable", "build_command": "lake build AISafetyAtlas.Sovereignty.Retargetable", "relationship": "RELATED", "scope_delta": { "summary": "Section 3 entire - definitions 3.1, 3.2, 3.3, 3.5, proposition 3.4 and theorem 3.6 - at print's own binders, WIDER on one axis throughout: print's symmetric group S_d is an arbitrary group here, and nothing below uses more than a group action. Definition A.6 is PARTIAL, carrying only the similarity half. ContainsCopies is definition A.7 with the involution clause dropped, which is wider; the printed form lives in the EUDetermined module and InvolutiveCopies.containsCopies records which is stronger. RELATED rather than EXACT because the module deliberately stops before power: proposition A.11 and the whole appendix D layer are absent.", "evidence": "docs/provenance/source-coverage-audit.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "smul_mem_orbitIn", "mostOrbit_of_orbitConditions", "IncreasingUnderJointPerm", "SupersetCopies", "mostOrbit_of_supersetCopies", "InvolutiveCopies", "InvolutiveCopies.containsCopies", "InvolutiveCopies.supersetCopies", "increasingUnderJointPerm_of_hide", "increasingUnderJointPerm_of_hide_invariant", "euProfile", "EUDetermined", "euProfile_smul", "EUDetermined.jointPermInvariant", "eu_determined_mostOrbit" ], "module": "AISafetyAtlas.Sovereignty.EUDetermined", "build_command": "lake build AISafetyAtlas.Sovereignty.EUDetermined", "relationship": "RELATED", "scope_delta": { "summary": "The chain from lemma B.7 to theorem A.13. MIXED. WIDER: item 1 of lemma B.7 is unguarded, which is a correction of print rather than a generalisation of it, since print's equation (25) applies item 1 where its printed guard is the negation of equation (29)'s conclusion, and print's lemma B.9 discharges it with no antecedent; lemma B.11 holds for an arbitrary invariant pairing, that being the single property print's equation (47) uses; theorem A.13 asks the looser B.8 hypothesis, bridged from print's A.7 by InvolutiveCopies.supersetCopies; B.8 and B.9 are stated over an arbitrary ordered G-set where print has a family of subsets of R^d. NARROWER, and costed in section 16: definition A.12's carrier is Finset V where print writes the power set of R^d. Not covered: lemma B.5, the expectation lift that is the one missing step to print's distribution-level statements, and proposition A.11's seven rationalities.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.MultiplyRetargetable.mostOrbit", "type": "NEW_PROOF", "source_declarations": [ "MultiplyRetargetable.mostOrbit", "SimplyRetargetable.mostOrbit" ], "application": "Cognitive sovereignty: a one-directional flippability in how a decision-maker can be retargeted forces a counting fact about every orbit of its parameter. Turner and Tadepalli, arXiv:2206.13477v2, theorem 3.6 and proposition 3.4." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.eu_determined_mostOrbit", "type": "NEW_PROOF", "source_declarations": [ "eu_determined_mostOrbit", "mostOrbit_of_orbitConditions", "mostOrbit_of_supersetCopies" ], "application": "Cognitive sovereignty: decision-makers determined by expected utility select the option set containing more copies at proportionally more parameters. Turner and Tadepalli, arXiv:2206.13477v2, theorem A.13, via the corrected form of lemma B.7." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "A decision-maker that can always be retargeted towards the larger set of options will favour it at most of its settings.", "use": "When arguing that a tendency comes from the shape of an option set rather than from any particular objective.", "title": "Retargetable decision-makers", "attribution": "Turner & Tadepalli" } }, { "id": "LAND-DEC-MDP-001", "name": "The rewardless Markov decision process, and the run it induces", "tags": [ "decision-theory" ], "notes": "GROUNDING. Turner and Tadepalli, arXiv:2206.13477v2, definition D.6, read from rendered page 35, which they attribute in turn to Turner et al. 2021. Graded in section 16 of docs/provenance/source-coverage-audit.md. CONTENT. Print: 'a rewardless MDP with finite state and action spaces S and A, and stochastic transition function T from S x A to Delta(S). We treat the discount rate gamma as a variable with domain [0,1].' MDP is that triple, curried, with PMF for Delta(S). WIDER, on two axes, both deliberate and both stated in the module: finiteness of S and A is dropped and no statement needs it, and there is no discount at all, because the consumer sums undiscounted over a finite horizon. WHY A CARRIER WITHOUT A REWARD FIELD. AISafetyAtlas.Wireheading.CRMDP makes the reward a field of an environment ranging over a class, not of the dynamics, which no carrier with a reward field can express. The module's earlier claim that three external Lean libraries had converged on this split was WRONG and was corrected on 2026-09-10: every external MDP structure opened that day puts the reward INSIDE the structure. The split is this repository's, taken from D.6. BEYOND PRINT. The run, the policy type, the observation map and their congruence and determinism lemmas are atlas-side; print states no run. The observation map is a parameter so a consumer chooses what a policy may read, which is what makes an environment and its complement indistinguishable to CRMDP. PMF rather than MeasureTheory.Kernel because the state type carries no measurable structure; moving to Kernel is a widening on a different axis and is not taken.", "original_source_refs": [ "atlas-ref-turner-tadepalli-2022-retargetable" ], "related_result_ids": [ "LAND-SOV-RETARGETABLE-001", "LAND-CRMDP-KNOW-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "MDP", "MDP.ofDet", "History", "Policy", "DetPolicy", "Policy.ofDet", "MDP.run", "MDP.stateAt", "MDP.historyUpTo", "MDP.run_congr_obs", "MDP.run_ofDet", "detRun", "detRun_congr_obs", "genRun", "genRun_id", "genRun_pmf" ], "module": "AISafetyAtlas.Decision.MDP", "build_command": "lake build AISafetyAtlas.Decision.MDP", "relationship": "RELATED", "scope_delta": { "summary": "Definition D.6's triple, WIDER on two axes print states and this carrier drops: D.6 requires S and A finite and carries a discount rate as a variable, and neither is present or needed here. RELATED rather than EXACT because the module is mostly BEYOND print - the run, the policy type, the observation map and their congruence lemmas are atlas-side, print stating no run at all.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Decision.MDP.run_congr_obs", "type": "NEW_PROOF", "source_declarations": [ "MDP", "MDP.run", "MDP.run_congr_obs" ], "application": "A rewardless dynamics carrier, so that a consumer whose reward ranges over a class of environments can supply the reward itself, and the lemma that makes the split pay: the run depends on the observation map only through what it observes, which is why an environment and its complement are indistinguishable to AISafetyAtlas.Wireheading.CRMDP. Turner and Tadepalli, arXiv:2206.13477v2, definition D.6 for the carrier; the run is not printed there." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "The dynamics of a decision process, carried separately from any reward defined over them.", "use": "When the reward is not a fixed part of the environment but something a consumer varies.", "title": "Rewardless Markov decision process", "attribution": "Turner & Tadepalli" } }, { "id": "LAND-GOODHART-SELECTION-001", "name": "Selection on a proxy: the inflated gap, and the region the link was never observed in", "tags": [ "agent-incentives", "decision-theory" ], "notes": "ROUTING FIRST, BECAUSE IT DETERMINES HOW TO READ THIS ROW. Manheim and Garrabrant number no theorem, proposition, lemma or corollary; their nine numbered items are generative 'Simple Model' specifications whose consequences are asserted in running prose. These two modules are therefore ATLAS-ORIGINAL work citing the paper for its MODEL, not reproductions of a printed result, exactly as docs/provenance/by037-by038-goodhart-campbell-plan.md directs. The model is print's; the theorems are this atlas's sharpening of print's prose into checkable statements. Graded statement by statement in section 17 of docs/provenance/source-coverage-audit.md, read in full from rendered pages 1 to 9. REGRESSIONAL, section 1. The carrier is a product of two copies of the line with goal, gap and proxy as coordinates and proxy = goal + gap, so the two-variable structure and the independence are properties of the space rather than assertions about it. gap_selection_ge is print's sentence 'when M is large, you can expect G to be predictably smaller than M', in product form: the selected gap mass is at least the unconditional mean gap times the selected mass, so nothing divides by a selected mass that may be zero and the statement survives a selection event of measure zero. gap_selection_gt adds the hypothesis strictness actually needs - a set of goal values of positive mass at which the threshold cuts the noise law - and the module carries a counterexample showing non-degenerate noise alone does NOT suffice, correcting an earlier informal claim. gap_selection_gaussianReal_gt is print's Gaussian case and asks the nonzero variance print leaves implicit. WIDER THAN PRINT: print writes normal noise, the theorems need only independence and integrability. EXTREMAL, section 2, both sub-variants. selected_disjoint_observed sharpens print's 'selection pressure moves the metric away from the region in which the relationship is most accurate' from a possibility into disjointness. FitsOn is print's 'approximately accurate in the initial region' with the tolerance explicit. regimeGoal is equation (3) at print's own comparisons. BEYOND PRINT: print states no error quantity, only that the relationship 'may be fundamentally different'; regime_prediction_error gives the exact size, and fits_underdetermined_off_observed and regime_underdetermines show that nothing observed inside the fitted region constrains the goal outside it, with print's own regime model as the witness rather than a spike planted at one state. NOT HERE, AND THIS IS WHERE THE PAPER'S CONTENT MOSTLY IS. Two thirds of the taxonomy: section 3 causal Goodhart, whose three intervention effects need an interventional layer the Goodhart cluster does not have - AISafetyAtlas.Causal.StructuralModel and AISafetyAtlas.Causal.Incentive are the substrate and BY-038 was parked pending exactly that; and section 4 adversarial Goodhart, needing a second actor. The sharpest uncovered claim is print's Campbell's Law row, equation (7), asserting that the correlation between the agent's goal and the agent's metric is zero over the full state set but positive on the subspace the regulator selects. Also not here: section 1's SECOND consequence, that the goal still rises under selection, which is a short step from tail_integral_ge applied to the goal coordinate and was simply never written.", "original_source_refs": [ "atlas-ref-manheim-garrabrant-2018-goodhart-variants" ], "related_result_ids": [ "BY-037", "BY-038" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "goal", "gap", "proxy", "selected", "measurableSet_selected", "measureReal_selected", "integral_gap_selected", "tail_integral_ge", "tail_integral_gt", "gap_selection_ge", "gap_selection_gt", "gap_selection_gaussianReal_gt" ], "module": "AISafetyAtlas.Goodhart.Regressional", "build_command": "lake build AISafetyAtlas.Goodhart.Regressional", "relationship": "RELATED", "scope_delta": { "summary": "RELATED, and deliberately not EXACT: print numbers no theorem, so there is no printed statement to be exact against. The MODEL is print's equation (1) at Same scope; the THEOREM is atlas-original, proving a consequence print asserts in prose without proof. WIDER on the noise - print writes normal, the theorems need only independence and integrability, with the Gaussian case recovered as a corollary that asks the nonzero variance print leaves implicit. The threshold is strict, following section 1's own sentence rather than the page-2 setup's non-strict one; the proof uses only that the selected slice lies above the threshold. NOT PROVED: print's second consequence, that the goal itself is higher under selection, which is a distinct claim from the gap growing.", "evidence": "docs/provenance/source-coverage-audit.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "selectedAt", "mem_selectedAt", "FitsOn", "selected_disjoint_observed", "fits_underdetermined_off_observed", "regimeGoal", "regimeGoal_eq_of_le", "regimeGoal_eq_of_lt", "regime_prediction_error", "regime_underdetermines", "regime_selected_disjoint_observed" ], "module": "AISafetyAtlas.Goodhart.Extremal", "build_command": "lake build AISafetyAtlas.Goodhart.Extremal", "relationship": "RELATED", "scope_delta": { "summary": "RELATED for the same reason: print numbers nothing here either. Equation (3) is rendered at Same scope, at print's own non-strict-below and strict-above comparisons. The two bolded sub-variant definitions are sharpened from possibility to a proved disjointness, which is WIDER than the prose. Equation (2) is only PARTIAL: as printed it names the residual and asserts nothing, so what the atlas carries is FitsOn, the relation the equation is about, rather than the equality itself. BEYOND print: the exact extrapolation error and the underdetermination of the goal off the observed region, neither of which print states in any form.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Goodhart.gap_selection_ge", "type": "NEW_PROOF", "source_declarations": [ "gap_selection_ge", "gap_selection_gt", "gap_selection_gaussianReal_gt", "tail_integral_ge" ], "application": "Specification gaming: selecting states by a proxy raises the expected proxy-goal gap, so an inexact metric diverges from the goal precisely where it is optimized. The model is Manheim and Garrabrant, arXiv:1803.04585v4, equation (1); the theorem is atlas-original, since that paper asserts the consequence in prose and proves nothing." }, { "atlas_declaration": "AISafetyAtlas.Goodhart.Extremal.fits_underdetermined_off_observed", "type": "NEW_PROOF", "source_declarations": [ "selected_disjoint_observed", "fits_underdetermined_off_observed", "regime_prediction_error", "regime_underdetermines" ], "application": "Specification gaming: once selection leaves the region where the proxy-goal link was fitted, the observations there do not constrain the goal at the selected states at all. Manheim and Garrabrant, arXiv:1803.04585v4, section 2, sharpened beyond what that paper states." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "Optimizing a proxy pushes the system into the tail, and into the region where the proxy was never checked against the goal.", "use": "When a measure fitted on ordinary conditions is about to be used as a target.", "title": "Selection on a proxy", "attribution": "after Manheim & Garrabrant" } }, { "id": "LAND-GOODHART-OVEROPT-001", "name": "Optimizing a proxy over some of the attributes floors the rest", "tags": [ "agent-incentives", "decision-theory" ], "notes": "GROUNDING. Zhuang and Hadfield-Menell, Consequences of Misaligned AI, NeurIPS 2020, Theorem 1, with the setup of sections 2.1.1 and 2.1.2. Statements read from rendered pages 3 to 8 of the published version, which defers all proofs to supplementary material not in the pinned file; Theorem 1's four-line proof is read from rendered page 5 of the arXiv preprint, pinned beside it. Graded in section 18 of docs/provenance/source-coverage-audit.md. CONTENT. zhuang_hadfield_menell_theorem_one is Theorem 1 at print's own binders - constraint function, lower bounds, proxy attribute set, proxy utility over the restriction, rate function, the integral representation of the optimization sequence, feasibility along it, Complete Optimization, and convergence - concluding that the omitted attributes sit at their lower bounds in the limit. proxyMax_unmentioned_eq_lowerBound is its core at an arbitrary attribute type with no finiteness, taking the maximizing property in place of the sequence. THE PRINTED THEOREM IS FALSE AND THIS IS THE REPAIR. Print says 'based on J < L attributes', bounding the count above and not below, and nothing in its setup excludes an empty proxy attribute set. At J = 0 the proxy is a constant, strictly increasing holds vacuously, every feasible state attains the supremum, a constant sequence converges, and the conclusion fails at any state above its floors. J.Nonempty is therefore a hypothesis, and the scope axis is closed by counterexample rather than owed. That this is oversight and not generality has two witnesses: print's own proof picks some feature in the proxy set to raise, and print states the hypothesis itself at Proposition 2, 'for any non-empty set of proxy attributes'. FOUR MORE READINGS PRINT LEAVES IMPLICIT. Its feasible set and its lower bounds are jointly unsatisfiable as written, so feasibleStates carries the box as a second constraint; print's closedness assumption is then derived by isClosed_feasibleStates rather than assumed. Print's proof justifies feasibility of its perturbation only by the constraint and never by the box, which is the only place the floor enters; the proof here supplies that step by taking the perturbation to be the whole distance to the floor. Complete Optimization equates two real numbers, so the proxy is bounded above on the feasible set; that is carried as an IsLUB hypothesis rather than as an indexed supremum, which would make the theorem false at an unbounded proxy. And the other half of J < L, that the proxy omits something, is not carried, which is a vacuous-true widening. BEYOND PRINT. le_of_unmentioned_eq_lowerBound and gap_le_of_unmentioned_eq_lowerBound say what the limit point costs the principal: it is the pointwise least feasible state carrying its own proxy attributes, so for any monotone true utility the proxy-goal gap there is at least its value at every feasible state with the same proxy attributes. Print draws no gap claim. NOT HERE. Theorem 1 is the paper's warm-up, not its result: Theorem 2 characterises when optimization is guaranteed costly, Theorem 3 says which attributes to put in the proxy, and both sit behind the paper's differential notion of sensitivity, which no module in this atlas has.", "original_source_refs": [ "atlas-ref-zhuang-hadfield-menell-2020-misaligned-ai" ], "related_result_ids": [ "LAND-GOODHART-SELECTION-001", "BY-037" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "StrictMonoCoord", "feasibleStates", "restrictAttrs", "isClosed_feasibleStates", "continuous_restrictAttrs", "proxyMax_unmentioned_eq_lowerBound", "le_of_unmentioned_eq_lowerBound", "monotone_of_strictMonoCoord", "gap_le_of_unmentioned_eq_lowerBound", "zhuang_hadfield_menell_theorem_one" ], "module": "AISafetyAtlas.Goodhart.Overoptimization", "build_command": "lake build AISafetyAtlas.Goodhart.Overoptimization", "relationship": "RELATED", "scope_delta": { "summary": "Theorem 1 at print's binders, NARROWER by exactly one hypothesis and the axis is CLOSED BY COUNTEREXAMPLE rather than owed: print's statement is false at an empty proxy attribute set, which its 'J < L' does not exclude, so J.Nonempty is a repair. Print itself states that hypothesis at Proposition 2 and its own proof needs it. WIDER on three axes: the true utility is asked only monotone where print asks continuous and strictly increasing; the core theorem is at an arbitrary attribute type with no finiteness and takes the maximizing property in place of the optimization sequence; and the half of 'J < L' saying the proxy omits something is dropped, which is vacuous-true. SAME on the setup, under the one reading that makes print consistent - print's feasible set and its lower bounds cannot both hold as written, so the box is a second constraint and print's closedness becomes derivable. RELATED rather than EXACT because of the added hypothesis and because two consequences are BEYOND print, which states no gap claim. Not covered: Theorems 2 and 3 and Propositions 1 to 5, all of which need the paper's differential sensitivity notion or its interactive game.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Goodhart.zhuang_hadfield_menell_theorem_one", "type": "NEW_PROOF", "source_declarations": [ "zhuang_hadfield_menell_theorem_one", "proxyMax_unmentioned_eq_lowerBound", "gap_le_of_unmentioned_eq_lowerBound" ], "application": "Specification gaming: when an agent optimizes a proxy defined over only some of the attributes a principal cares about, the attributes it omits are driven to their lower bounds. Zhuang and Hadfield-Menell, NeurIPS 2020, Theorem 1, with the nonemptiness hypothesis print omits and its own Proposition 2 supplies." } ] }, "public": { "group": "Preferences, rewards and incentives", "summary": "An objective that mentions only some of what you care about drives the rest to the floor.", "use": "When choosing which attributes to write into a metric an agent will optimize.", "title": "Overoptimization floors the unmentioned", "attribution": "Zhuang & Hadfield-Menell" } }, { "id": "LAND-SOV-POWER-001", "name": "Peleg's game-form layer: alpha-effectivity, and the conditions it satisfies for free", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. Peleg, Effectivity functions, game forms, games, and rights, Social Choice and Welfare 15: 67-80 (1998), pinned as peleg1997.pdf, section 3 read from rendered journal pages 72 and 73 and section 2 from 68 and 69. Graded in section 21 of docs/provenance/source-coverage-audit.md, the first section this source has had, and the first registry row for either module. CONTENT. GameForm is Definition 3.2 without the constitution, Forces is Definition 3.3's alpha-effectivity at print's quantifier order, effectivity is Definition 3.3's family, and Represents is the shape of Definition 3.4. WIDER: four conditions print imposes on an effectivity function are derived here rather than assumed - forces_univ is print's condition that every coalition is effective for the whole space, Forces.mono is (3.3), Forces.mono_coalition is (3.5), and forces_superadditive is Definition 3.1. Print states all four as properties that may fail, and Theorem 3.5 makes two of them the criterion for representability; for the alpha-effectivity function of a game form all four are theorems. Non-emptiness of each strategy set is print's own Definition 3.2 requirement and is carried as an instance hypothesis, not as an added one. NARROWER, AND SAID SO UNTIL 2026-09-20: Represents was Partial, not Yes, because print's Definition 3.4 is a condition on a CONSTITUTION and the development had none, so Represents put a second game form's own alpha-effectivity function where print puts the induced one - a strictly weaker question, since a game form's alpha-EF is representable by construction. See the rights-layer paragraph at the end of this note for how that was closed. NOT HERE: the society, the constitution, legality, surjectivity of the outcome function, and Theorem 3.5 itself. Both of Theorem 3.5's conditions are theorems here, so the missing content is exactly the construction of a game form from a prescribed effectivity function. ATLAS-SIDE, GRADED Beyond AND NOT AGAINST THIS SOURCE: the mandate family, RetainsFamily and the retention comparison in Sovereignty.Mandate, and the separations in Sovereignty.Separations. Per docs/provenance/cognitive-sovereignty-obligation.md section 8 item 1 the mandate vocabulary is unpublished and is graded as interpretation. ONE DEFECT OF PRINT: Definition 3.3 prints the family as sets B contained in the coalition S; it is a typo for B contained in the social states A. SECOND SOURCE (2026-09-11). The ActualPower layer renders Chen, Ju and Agotnes, arXiv:2607.10567v1, Definitions 5 and 9, graded in section 23 of the coverage audit. GameForm is that paper's SID frame class - serial, independent, deterministic - which is why superadditivity is a theorem here; print's own carrier is AISafetyAtlas.Sovereignty.ConcurrentGameFrame as of 2026-09-20, gameFrame is the embedding, and AISafetyAtlas.Examples.Sovereignty.exists_gcgf_not_superadditive is a frame where superadditivity FAILS. Two provenance defects in this module's docstrings were found and fixed by that grading: a quoted objection that is in the authors' two-agent companion paper rather than in the cited one, and credited there to a third source; and an inverted reading of print's caution that the eight frame labels do not record which conditions must fail. SOV-1 IS STATED, 2026-09-13, AND THE ITEM THAT BLOCKED IT WAS ANSWERED BY THE MAINTAINER RATHER THAN BY A PROOF. docs/provenance/cognitive-sovereignty-obligation.md section 8 item 5 left exactly one thing open: in the delegated game, which coalition acts for the principal. The ruling is that this is not a free modelling choice - it is fixed by the delegate's alignment. An aligned delegate acts for the principal, so the coalition is the pair; an unaligned one does not, so the coalition is the principal alone and the delegate's strategies join the complement the guarantee has to survive. principalCoalition carries that rule as a set-valued function of an ARBITRARY alignment proposition - an insert rather than an if, so no decidability instance is needed and the proposition may be one nobody in the model can evaluate - with principalCoalition_of_aligned and principalCoalition_of_unaligned the two computations, sov1_of_aligned and sov1_of_unaligned the two readings, and SOV1 is RetainsFamily at it. WHAT THE RULE BUYS. sov1_of_retainsFamily_singleton says discharging SOV-1 as though the delegate were unaligned discharges it at EVERY alignment: refusing to assume alignment costs nothing, and assuming it is what costs something. NOT VACUOUS AND NOT A RELABELLING. AISafetyAtlas.Examples.Sovereignty.sov1_turns_on_alignment exhibits one principal, one mandate and one pair of games on which SOV-1 holds at the aligned setting and fails at the unaligned one, built from the witnesses that were already there: sov1_aligned runs on retainsWith_delegated and not_sov1_unaligned on retainsAgainst_undelegated together with not_retainsAgainst_delegated. READING 3 IS PARALLEL, NOT AN INSTANCE. A pinned delegate policy is a restriction of a strategy set and not a coalition, so RetainsUnderFamily is stated separately, with retainsUnder_of_retainsUnderFamily the single-target specialization and RetainsUnderFamily.mono_mandate its monotonicity; AISafetyAtlas.Examples.Sovereignty.retainsUnderFamily_delegated_zero against AISafetyAtlas.Examples.Sovereignty.not_retainsUnderFamily_delegated_two carries reading 3's policy-dependence to the family level. ATLAS-SIDE THROUGHOUT: the alignment rule is the maintainer's and would grade Beyond, not against Peleg. NOT HERE: nothing decides whether a given delegate is aligned and nothing supplies a criterion for it, exactly as the mandate family is an input rather than a theorem. THE RIGHTS LAYER LANDED 2026-09-20, AND IT CLOSED THE SECTION'S LAST NARROWING. AISafetyAtlas.Sovereignty.Rights builds what the row above had recorded as NOT HERE. Constitution is Peleg's Definition 2.4, the triple of rights, assignment and access correspondence, with print's rho as an arbitrary type rather than a finite set. Constitution.induced is (3.1). Constitution.Standing is the standing assumption on journal page 69, carried as a predicate rather than as structure fields because Definition 3.4 needs none of it. MonotoneAssign, MonotoneAlternatives, MonotoneRights and MonotoneCoalitions are (3.2) to (3.5) as predicates a constitution may fail, which is how print states them; induced_mono is the one statement needing three of them at once. Represents is Definition 3.4 at print's right-hand side. THE OLD STAND-IN, AND EXACTLY HOW IT WAS WEAKER: the game-form-to-game-form comparison survives as EffectivityEq, and effectivityEq_iff_represents_ofGameForm identifies it as representation against Constitution.ofGameForm -- a constitution read off one of the two game forms, which represents_ofGameForm shows every game form passes against itself. THREE WIDENINGS, NONE A STRENGTHENING: print restricts Definition 3.4 to LEGAL game forms and leaves legality informal, the equality uses none of it; print's rho is finite and this is not; and print assumes the outcome function onto from Definition 3.3 onward, where Represents.surjective_outcome DERIVES it from representation of a constitution obeying the standing assumptions. THEOREM 3.5 IS NOW PARTIAL RATHER THAN NO. Represents.superadditive is necessity of its condition (ii). Necessity of (i) as printed is NOT available and that is a fact about print's quantifier: (3.3) quantifies gamma over every theta, while a representation constrains only the diagonal gamma(S, alpha(S)), which print's own page 70 says is all that enters the analysis. Represents.upwardClosed is (3.3) on the diagonal and AISafetyAtlas.Examples.Sovereignty.represented_not_monotoneAlternatives is a represented constitution that fails (3.3) off it, so the shortfall is exhibited and not asserted. Surjectivity being derived rather than assumed also closes print's EF condition (ii): Playability.effectivity_univ_eq_nonempty. What remains is the sufficiency direction, print's appendix construction, costed in section 21. A SECOND LOOSENESS OF PRINT, and it is looseness rather than a typo: Example 2.3 lists gamma(1, rho_1) and gamma(2, rho_1) as two sets each -- the MINIMAL attainable sets -- while writing gamma(N, rho_1) = 2^A minus the empty set in full, in the same sentence, so within one example one coalition's family is given entire and two by generators. Remark 2.2 on the same page supplies the reading and Example 2.10 writes the closure out with its B-plus notation. Taken as whole families the listing fails print's own (3.3), and Examples.Sovereignty.not_represents_of_induced_eq_listed shows that no game form represents ANY constitution whose induced family at member 1 is the listing -- which would contradict Example 3.6. Examples.Sovereignty.not_monotoneAlternatives_listedShirt is the (3.3) failure itself. Relatedly, print's E(empty) = A on pages 72 and 73 is not a typo but print's own abbreviation, declared on page 69: 'henceforth, we shall denote a singleton {a} by a'. WITNESSES, IN THE SAME COMMIT: AISafetyAtlas.Examples.Sovereignty.Rights renders Example 2.3 as shirtConstitution with its standing assumptions and all three monotonicity conditions proved, Example 3.6 as represents_shirtGame, and Example 3.7 as not_represents_sequentialGame -- print's own positive and negative pair, at print's own strategy f(x) = x. NOT HERE, STILL: legality, Remark 2.5's equal-treatment condition, Examples 2.7, 2.10 and 3.8, Theorem 3.5's construction, and sections 4 and 5.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights", "chen-ju-agotnes-2026-actual-alpha-powers" ], "related_result_ids": [ "LAND-SOV-RETARGETABLE-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Constitution", "Constitution.induced", "Constitution.Standing", "Constitution.induced_empty", "Constitution.MonotoneAssign", "Constitution.MonotoneAlternatives", "Constitution.MonotoneRights", "Constitution.MonotoneCoalitions", "Constitution.induced_mono", "Superadditive", "UpwardClosed", "Represents", "Constitution.ofGameForm", "represents_ofGameForm", "EffectivityEq", "EffectivityEq.refl", "EffectivityEq.symm", "EffectivityEq.trans", "effectivityEq_iff_represents_ofGameForm", "Represents.superadditive", "Represents.upwardClosed", "Represents.surjective_outcome" ], "module": "AISafetyAtlas.Sovereignty.Rights", "build_command": "lake build AISafetyAtlas.Sovereignty.Rights", "relationship": "RELATED", "scope_delta": { "summary": "Peleg's rights layer at print's own statements, built 2026-09-20. Constitution is Definition 2.4, induced is (3.1), Standing is the standing assumption on journal page 69, MonotoneAssign / MonotoneAlternatives / MonotoneRights / MonotoneCoalitions are (3.2) to (3.5) as predicates a constitution may fail, and Represents is Definition 3.4 against the constitution - which closes the one Narrower cell section 21 carried. WIDER on three axes, none a strengthening: print restricts Definition 3.4 to LEGAL game forms and the equality uses no part of legality; print's set of rights is finite and the type R here is arbitrary; and print assumes the outcome function onto from Definition 3.3 onward, where Represents.surjective_outcome derives it instead. NARROWER on Theorem 3.5, which is Partial: Represents.superadditive is necessity of condition (ii), Represents.upwardClosed is condition (i) restricted to the diagonal that a representation can see, and the sufficiency direction - print's appendix construction - is not built and is costed in section 21. RELATED rather than EXACT because Theorem 3.5 is the row's headline result and only half of its necessity direction is here.", "evidence": "docs/provenance/source-coverage-audit.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "retainsFamily_of_effectivityEq", "effectivityEq_of_retainsFamily_univ", "RetainsFamily", "retainsFamily_of_represents", "RetainsFamily.trans", "RetainsFamily.mono_mandate", "RetainsFamily.mono_coalition", "principalCoalition", "principalCoalition_of_aligned", "principalCoalition_of_unaligned", "singleton_subset_principalCoalition", "SOV1", "sov1_of_aligned", "sov1_of_unaligned", "sov1_of_retainsFamily_singleton", "SOV1.mono_mandate", "RetainsUnderFamily", "retainsUnder_of_retainsUnderFamily", "RetainsUnderFamily.mono_mandate", "retainsWith_of_retainsFamily", "retainsAgainst_of_retainsFamily" ], "module": "AISafetyAtlas.Sovereignty.Mandate", "build_command": "lake build AISafetyAtlas.Sovereignty.Mandate", "relationship": "RELATED", "scope_delta": { "summary": "Peleg's Definition 3.4 moved to AISafetyAtlas.Sovereignty.Rights on 2026-09-20, where it is stated against a constitution as print states it; this module's game-form-to-game-form stand-in survives there as EffectivityEq. What remains here is atlas-side interpretation throughout - the mandate family, RetainsFamily, the three readings and SOV-1 - and is graded Beyond, not against Peleg. retainsFamily_of_represents is now the constitution-side statement: two game forms representing the same constitution retain every mandate across the move between them. RELATED rather than EXACT because nothing printed states any of it. RetainsFamily.trans (2026-09-11) is transitivity of retention at a fixed mandate family; it is the hypothesis the steering path in LAND-SOV-STEERING-001 removes.", "evidence": "docs/provenance/source-coverage-audit.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "GameForm", "Forces", "effectivity", "Forces.mono", "forces_univ", "Forces.mono_coalition", "forces_superadditive", "Forces.inter_nonempty_of_disjoint", "iInter_nonempty_of_pairwiseDisjoint", "forces_of_forces_singletons", "dominated_forces_univ", "minCard_cannot_separate", "ActualPower", "ActualPower.forces", "pinned_actualPower_singleton", "pinned_not_actualPower_pair", "pinned_forces_pair" ], "module": "AISafetyAtlas.Sovereignty.Separations", "build_command": "lake build AISafetyAtlas.Sovereignty.Separations", "relationship": "RELATED", "scope_delta": { "summary": "Section 3's definitions at print's binders, WIDER on the four conditions print imposes and this module derives (forces_univ, Forces.mono, Forces.mono_coalition, forces_superadditive), and wider again in allowing the empty coalition and dropping print's surjectivity of the outcome function. Print's conditions (i) and (ii) on an effectivity function are therefore not recovered, and grade Partial and No. The separations themselves answer questions this source does not ask and are graded Beyond. SECOND SOURCE, added 2026-09-11: ActualPower is Definition 5's actual effectivity function from Chen, Ju and Agotnes, arXiv:2607.10567v1, graded in section 23; GameForm is that paper's SID frame class, and pinned_not_actualPower_pair exhibits the alpha-versus-actual separation the paper motivates its notion by. Print's own carrier is AISafetyAtlas.Sovereignty.ConcurrentGameFrame from 2026-09-20 and gameFrame places this module at the SID corner; the state set there is the outcome type, not a point.", "evidence": "docs/provenance/source-coverage-audit.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "ActionFrame", "restrictJA", "jointUnion", "SafeFor", "TightFor", "AlphaPower", "ActualPowerAt", "alphaEff", "actualEff", "actualPowerAt_iff_out_eq", "AlphaPower.mono", "ActualPowerAt.alphaPower", "IsGCGF", "IsGCGF.mem_out_iff", "IsGCGF.mem_av_iff", "IsGCGF.out_anti_coalition", "IsGCGF.out_eq_iUnion_av", "IsGCGF.exists_av_univ_restrict", "IsGCGF.alphaPower_mono_coalition", "IsGCGF.not_alphaPower_empty", "ActionFrame.ofGrand", "ofGrand_out_univ", "ofGrand_isGCGF", "Serial", "Independent", "Deterministic", "FrameClass", "card_frameClass", "IsFrame", "isFrame_mono", "isFrame_of_sid", "gameFrame", "gameOut_eq_outcomesOf", "gameFrame_isGCGF", "alphaPower_gameFrame", "alphaEff_gameFrame", "actualPowerAt_gameFrame", "gameFrame_serial", "gameFrame_independent", "gameFrame_deterministic", "gameFrame_isFrame_sid" ], "module": "AISafetyAtlas.Sovereignty.ConcurrentGameFrame", "build_command": "lake build AISafetyAtlas.Sovereignty.ConcurrentGameFrame", "relationship": "RELATED", "scope_delta": { "summary": "Chen, Ju and Agotnes arXiv:2607.10567v1 at print's own carrier, built 2026-09-20 to close the four cells section 23 of docs/provenance/source-coverage-audit.md had recorded as Narrower. ActionFrame is Definition 1: a state set, an action set, and per coalition an availability function and an outcome function into SETS of states. AlphaPower and ActualPowerAt are Definition 5's two effectivity functions with print's safe and tight as SafeFor and TightFor; actualPowerAt_iff_out_eq is print's own 'equivalently'. IsGCGF is Definition 8's two conditions and ActionFrame.ofGrand makes print's next sentence - that they DETERMINE the coalition-level functions from the grand coalition's outcome function - a construction print never builds. Serial, Independent and Deterministic are Definition 9, FrameClass and IsFrame are its eight labels with card_frameClass the count and isFrame_mono print's caution that the labels record what is imposed and not what must fail. IsGCGF.out_anti_coalition and IsGCGF.out_eq_iUnion_av are Fact 1's two items. WIDER throughout on three dropped hypotheses: print's agent set is finite, its state set nonempty and its action set nonempty, and none of the three is carried. WIDER once beyond the carrier: IsGCGF.alphaPower_mono_coalition needs NO seriality, where the game-form version Forces.mono_coalition carries inhabitance of every strategy type; the outcome-driven availability condition supplies the extension instead. BEYOND: IsGCGF.not_alphaPower_empty, that no coalition is ever effective for the empty set, which print does not state and which is why superadditivity is not a theorem here. THE BRIDGE, AND A CORRECTION IT FORCED: gameFrame places AISafetyAtlas.Sovereignty.Separations at print's SID corner, and its state set is the OUTCOME TYPE with state-independent dynamics. Section 23 and the Separations header both said 'at a single state' until this module was built; with a one-point state set every alpha power is trivial, so the claim was false and unfalsifiable at once, because nothing in the atlas could state print's carrier. alphaPower_gameFrame, alphaEff_gameFrame and actualPowerAt_gameFrame identify Forces, effectivity and ActualPower as the induced frame's alpha and actual effectivity functions, and gameFrame_isFrame_sid places it. AISafetyAtlas.Sovereignty.AATS is a DIFFERENT corner and no bridge is built: Wooldridge and van der Hoek's system has states but its availability is per agent and its transition a partial function, so independence and determinism are built in there as they are in a game form. NOT HERE: the neighborhood-frame layer, representability, and every numbered theorem - the representation theorems are what the paper is for and this module claims nothing about representation. WITNESSES in AISafetyAtlas.Examples.Sovereignty.ConcurrentGameFrame: endFrame_not_serial is print's motivating game that ends, coinFrame_not_deterministic a grand-coalition action with two successors, clashFrame_not_independent two agents whose individually available actions are not jointly available, and exists_gcgf_not_superadditive the consequence - superadditivity, a theorem about every game form, FAILS at a general concurrent game frame.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.forces_superadditive", "type": "NEW_PROOF", "source_declarations": [ "forces_superadditive", "Forces.mono_coalition", "Forces.mono", "forces_univ" ], "application": "Cognitive sovereignty: the four conditions Peleg imposes on an effectivity function hold automatically for the alpha-effectivity function of a game form, so none of them distinguishes a dominated principal from an undominated one. Peleg 1998, Definition 3.1 and conditions (3.3) and (3.5)." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.retainsFamily_of_represents", "type": "NEW_PROOF", "source_declarations": [ "Represents", "Constitution", "Constitution.induced", "retainsFamily_of_represents", "effectivityEq_of_retainsFamily_univ" ], "application": "Cognitive sovereignty: mandate retention across a delegation sits strictly below Peleg's Definition 3.4 and meets it at the extremes. Definition 3.4 is now Represents, against the constitution rather than against a second game form; the mandate family is atlas-side interpretation, not Peleg's." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.sov1_of_retainsFamily_singleton", "type": "NEW_PROOF", "source_declarations": [ "principalCoalition", "SOV1", "sov1_of_retainsFamily_singleton" ], "application": "Cognitive sovereignty: SOV-1 with the delegated coalition fixed by the delegate's alignment rather than chosen. Atlas-side interpretation of an open item in docs/provenance/cognitive-sovereignty-obligation.md, not anything Peleg states." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.Represents.surjective_outcome", "type": "NEW_PROOF", "source_declarations": [ "Constitution", "Constitution.induced", "Constitution.Standing", "Represents", "represents_ofGameForm", "Represents.superadditive", "Represents.upwardClosed", "Represents.surjective_outcome" ], "application": "Cognitive sovereignty: whether a delegated mechanism distributes power the way the constitution says it should, rather than merely the way some other mechanism does. Peleg 1998, Definition 2.4, equation (3.1) and Definition 3.4." } ] }, "public": { "group": "What an observer can recover", "title": "Power in a game form", "summary": "What a group can guarantee no matter what everyone else does, and which properties of that come for free.", "use": "When asking whether delegating a decision costs the principal anything it could previously guarantee.", "attribution": "Peleg; the mandate layer is atlas-side interpretation" } }, { "id": "LAND-SOV-INDEPENDENCE-001", "name": "The logical space of freedom, and the corner that cannot escape", "tags": [ "multi-agent", "oversight" ], "notes": "GROUNDING. List and Valentini, Freedom as Independence, Ethics 126(4): 1043-1074 (2016), sections II and III read from rendered journal pages 1046 and 1047; and Carter and Shnayderman, The Impossibility of 'Freedom as Independence', Political Studies Review (2018), OnlineFirst, both displayed theses read from the article's own pages 4 and 5. Graded jointly in section 22 of docs/provenance/source-coverage-audit.md, the first section either source has had and the first registry row this module has had. CONTENT. Setting is print's frame with the relevant-worlds class as a parameter, because print says it is one, and with the actual world a member of it, because print's robustness is 'over and above the actual world'. Counts is the moralization question's answer, worlds is the robustness question's answer, and Free is print's two-by-two. LiberalFree, MoralizedLiberalFree, IndependenceFree and RepublicanFree are print's four cases under print's own attributions. constrains and permitted are separate fields because print's footnote 8 forbids moralizing the notion of relevance. BOTH THESES, AS CONDITIONALS. not_independenceFree_of_universal_threat and not_republicanFree_of_impermissible_threat are Carter and Shnayderman's two displayed theses. Their antecedents are claims about politics that print supplies and the atlas does not assert. The asymmetry between the two hypotheses - one needs an impermissible constraint, the other needs only a constraint - IS print's argument, namely that List and Valentini drop the arbitrariness requirement. INDEPENDENCE OF THE TWO DIMENSIONS, CLOSED 2026-09-11: print asserts the moralization and robustness dimensions are independent, hence that the matrix has four cells. Examples.Sovereignty.four_corners_distinct is a Setting whose four conceptions are four different sets of free actions, so the claim is now witnessed rather than carried by the two implications alone. NARROWER, AND PRINT SAYS SO: print states both dimensions can be refined and made nonbinary and that each case is a family; Free takes two Bool parameters. Print's claim that all four families are modal is not visible here either, since it is a claim about what a constraint is and constrains is a primitive. NOT HERE: the functional-role desideratum, List and Valentini's positive case, and Carter and Shnayderman's Kramer probability diagnosis.", "original_source_refs": [ "list-valentini-2016-freedom-as-independence", "carter-shnayderman-2018-impossibility" ], "related_result_ids": [ "LAND-SOV-POWER-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Setting", "Counts", "worlds", "Free", "LiberalFree", "MoralizedLiberalFree", "IndependenceFree", "RepublicanFree", "free_of_free_not_moralized", "free_of_free_robust", "republicanFree_of_independenceFree", "liberalFree_of_independenceFree", "not_independenceFree_of_universal_threat", "not_republicanFree_of_impermissible_threat" ], "module": "AISafetyAtlas.Sovereignty.Independence", "build_command": "lake build AISafetyAtlas.Sovereignty.Independence", "relationship": "RELATED", "scope_delta": { "summary": "Print's definition scheme, both named questions and all four numbered cases at print's own parameters, plus both of Carter and Shnayderman's displayed theses as conditionals on the political premises print supplies. NARROWER on one axis print itself names: the two dimensions are Bool here, where print says they can be refined and made nonbinary. The two dimensions are independent: four_corners_distinct exhibits a Setting in which the four conceptions are four different sets of free actions. RELATED rather than EXACT because neither source numbers a theorem and the module renders theses and a taxonomy, not proofs.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.Setting.not_independenceFree_of_universal_threat", "type": "NEW_PROOF", "source_declarations": [ "not_independenceFree_of_universal_threat", "not_republicanFree_of_impermissible_threat" ], "application": "Cognitive sovereignty: if a robust, non-moralized reading of freedom is taken over sheer possibility, a universally available threat makes nobody free to do anything. Carter and Shnayderman (2018), both displayed theses, over List and Valentini's (2016) case 3 and case 4." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.Setting.republicanFree_of_independenceFree", "type": "NEW_PROOF", "source_declarations": [ "republicanFree_of_independenceFree", "liberalFree_of_independenceFree", "free_of_free_not_moralized", "free_of_free_robust" ], "application": "Cognitive sovereignty: the moralization and robustness questions are independent parameters, and the resulting four conceptions are ordered - moralizing only adds freedom, robustness only removes it, so freedom as independence implies both liberal and republican freedom and is the strongest of the four. List and Valentini (2016), sections II and III." } ] }, "public": { "group": "What an observer can recover", "title": "Four conceptions of freedom", "summary": "Whether you are free can depend on other possible worlds, and on whether the obstacle was permitted - two independent choices, four answers.", "use": "When a claim about autonomy has to say which notion of freedom it is using.", "attribution": "List & Valentini; Carter & Shnayderman" } }, { "id": "LAND-SOV-STEERING-001", "name": "Stepwise retention is not authorship: locally safe steps that lose the original mandate", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. No printed theorem is graded here, and this row claims none. The substrate is Peleg's game-form layer (GameForm, Forces, effectivity in AISafetyAtlas.Sovereignty.Separations, hosted by LAND-SOV-POWER-001, graded in section 21 of docs/provenance/source-coverage-audit.md) together with the atlas-side mandate layer (RetainsFamily in AISafetyAtlas.Sovereignty.Mandate). The concept this module renders is the HEC2026 keynote cut of cognitive sovereignty -- the capacity to remain the AUTHOR of your own memory, goals, judgment and action under AI mediation -- which is unpublished and therefore uncitable; docs/provenance/cognitive-sovereignty-obligation.md section 1 records all three published definitions and why none is gradeable as written. Everything in this row is atlas-side interpretation and would grade Beyond, not against Peleg. CONTENT. SteeringPath is a path of n + 1 game forms with a mandate family attached to each and a StepKind for each of the n adjacent steps, at a fixed principal coalition C: this is gradual steering of one principal, not a change of who occupies the slot. StepSafe i is retention of the LATER mandate across step i -- the step looks safe against what the principal now treats as its own mandate. AdjacentSafe is that for every step, OriginalLost is failure of retention from G 0 to G n against mandate 0, and IsSteering is their conjunction. THE CLAIM, AND WHY IT IS NOT VACUOUS. RetainsFamily.trans says retention composes at a FIXED family, so IsSteering is inhabited only because the family moves. AISafetyAtlas.Examples.Sovereignty.steer_isSteering is the witness: one principal (false) and one assistant (true), outcomes Fin 3 with 0 the original goal, 1 a substitute and 2 a disaster. Memory coarsens the mandate from {0} to {0, 1} with the game unchanged; Compass (reframed) reads a choice of 0 as 1, so the original option is still on the button and no longer means what it did; Engine (executed) hands naming to the assistant, which maps disaster to the substitute. steer_step0, steer_step1 and steer_step2 are the three adjacent retentions, steer_originalLost is the failure across the composition, and steer_retains_coarsened shows the path DOES retain the coarsened family end to end -- so the phenomenon is the moving mandate and not a failure of transitivity. NOT HERE. Nothing says a steering path is reachable by any particular assistant, nothing quantifies how much coarsening a step may do (SteeringPath requires since 2026-10-05 that each mandate coarsens the one before, so a path cannot look safe by emptying the mandate, and that a Memory step keeps the game while Compass and Engine keep the mandate; the closure audit had found later mandates could be empty and kind unread; Compass and Engine are still not told apart), and the five verbs of the keynote cut (preserve, inspect, contest, redirect, recover) are named in the module docstring as interpretation and are not formalized.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights" ], "related_result_ids": [ "LAND-SOV-POWER-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "StepKind", "SteeringPath", "StepSafe", "AdjacentSafe", "OriginalLost", "IsSteering" ], "module": "AISafetyAtlas.Sovereignty.Steering", "build_command": "lake build AISafetyAtlas.Sovereignty.Steering", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side interpretation throughout: no source states SteeringPath, AdjacentSafe, OriginalLost or IsSteering, so there is no printed statement to be Same, Wider or Narrower than, and every claim in this module would be graded Beyond. The substrate is Peleg's alpha-effectivity (Definitions 3.2 and 3.3, hosted by LAND-SOV-POWER-001) reached through RetainsFamily, which is itself atlas-side. RELATED, not EXACT: the row records what the atlas proves about a concept the cited source does not discuss. The worked path lives in AISafetyAtlas.Examples.Sovereignty.Steering (steer, steer_step0, steer_step1, steer_step2, steer_adjacentSafe, steer_originalLost, steer_isSteering, steer_retains_coarsened) and is reached from the root facade, so its theorems are inside the axiom audit.", "evidence": "docs/provenance/cognitive-sovereignty-obligation.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Examples.Sovereignty.steer_isSteering", "type": "NEW_PROOF", "source_declarations": [ "steer_isSteering", "steer_adjacentSafe", "steer_originalLost", "steer_retains_coarsened", "RetainsFamily.trans" ], "application": "Cognitive sovereignty: a delegation path can retain the principal's CURRENT mandate at every step and still not retain the mandate it started with. Since retention composes at a fixed mandate family (RetainsFamily.trans), the only way to lose authorship through locally safe steps is for the mandate itself to move -- which is what a Memory step does. Atlas-side interpretation of the HEC2026 keynote cut; not a claim of any pinned source." } ] }, "public": { "group": "Limits of control and regulation", "title": "Losing the plot one safe step at a time", "summary": "Every step keeps what you currently ask for, and the thing you originally asked for is gone by the end.", "use": "When a delegation is reviewed step by step and each review asks only whether this step is acceptable now.", "attribution": "Peleg's game-form layer; the steering reading is atlas-side interpretation of an unpublished definition" } }, { "id": "LAND-SOV-SERVICE-001", "name": "Safety without service: a delegate that refuses everything keeps every guarantee it already satisfies", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. No printed theorem is graded here, and this row claims none. The source is an unpublished internal proposal, A Formal System of Power, Sovereignty, and Cognitive Sovereignty, pinned by sha256 in docs/provenance/formal-power-proposal-triage.md together with its executable finite model finite_models_power.py. Because it is unpublished it is not in docs/provenance/source-coverage-audit.md and nothing here is coverage. The substrate is Peleg's game-form layer (GameForm, Forces, effectivity in AISafetyAtlas.Sovereignty.Separations, hosted by LAND-SOV-POWER-001) and the atlas-side mandate layer (RetainsFamily in AISafetyAtlas.Sovereignty.Mandate). CONTENT. The proposal's result D8 says safety without service is vacuous sovereignty: a controller that refuses every input preserves no unauthorized change while failing every correction or task-completion demand, so invariant preservation does not imply the service-based sovereignty definition. Inert is that controller at the level of a game form -- nothing anyone does changes the outcome. Inert.forces_iff computes its whole effectivity family: the sets containing the still point, at every coalition including the empty one. Demandwise is the proposal's section 5.3 demandwise sovereignty at the sure reading, which is exactly what the proposal's own Python computes as all(self.can(q) for q in protected). retainsFamily_and_not_demandwise is D8. WHAT IS WIDER THAN PRINT. inert_retainsFamily_iff characterizes the class instead of exhibiting one controller: a refusing delegate retains a protected family if and only if its still point lies in every protected set the baseline actually delivered, which is the proposal's own E(G0) intersect P. Separating is the proposal's section 5.2 nonvacuity condition -- no single outcome meets the whole catalogue -- and separating_not_demandwise_of_inert defeats EVERY refusing delegate at once rather than one pinned to a chosen point. WHAT IS NARROWER THAN PRINT. D8 is stated in the proposal's section 6, a finite deterministic safety game with a controlled-invariant kernel computed by iterating a controllable predecessor. This module has no transition relation, no trajectory and no time index; it renders D8's mechanism in the one-shot effectivity frame. D1 to D3 -- the safety kernel, the recovery attractor, budget monotonicity -- are not built. NOT VACUOUS. AISafetyAtlas.Examples.Sovereignty.Service carries the witnesses at Fin 3, read as 0 nothing changed, 1 the authorized correction applied, 2 an unauthorized change: d8 is the conjunction at refusing = pinnedForm 0, serving_demandwise shows the catalogue is satisfiable by principalDecides so the failure is about the delegate and not an impossible specification, separating_rival exhibits a separating catalogue, no_inert_demandwise_rival defeats every inert delegate at it, and serving_separating shows even that stronger condition is met by a real game form.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights", "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-POWER-001", "LAND-SOV-STEERING-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Inert", "Demandwise", "Separating", "inert_retainsFamily_iff", "retainsFamily_and_not_demandwise", "separating_not_demandwise_of_inert" ], "module": "AISafetyAtlas.Sovereignty.Service", "build_command": "lake build AISafetyAtlas.Sovereignty.Service", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of one result of an unpublished proposal, so there is no published printed statement to be Same, Wider or Narrower than and every claim would be graded Beyond against any pinned source. Against the proposal itself the rendering is Mixed: wider in that inert_retainsFamily_iff characterizes exactly which protected families are vacuously retainable and Separating rules out every refusing delegate rather than one, narrower in that the proposal states D8 in a dynamic safety game with a controlled-invariant kernel and nothing here has dynamics. RELATED, not EXACT. Two of the proposal's definitions were checked against the document and are already this repository's: its section 5.4 relative retention is RetainsFamily, and its section 5.3 demandwise sovereignty is Demandwise. Witnesses live in AISafetyAtlas.Examples.Sovereignty.Service and are reached from the root facade, so they are inside the axiom audit.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "Audit", "refusal_passes_safety", "refusal_fails_catalogue", "safety_suite_admits_a_refusal", "isRefusalHole", "admitsRefusal", "exists_refusal_hole_of_admitsRefusal", "not_exists_refusal_hole_of_admitsRefusal_eq_false" ], "module": "AISafetyAtlas.Sovereignty.Refusal", "build_command": "lake build AISafetyAtlas.Sovereignty.Refusal", "scope_delta": { "summary": "BRIDGE module with its executable half, backing atlas-check kind 'refusal'. Unlike the other checkers this one is EXACT rather than one-sided: both the positive and the negative verdict carry an agreement theorem, because the question is a finite search over outcomes with no quantifier over systems in it. Witnessed at a three-outcome assistant audit in AISafetyAtlas.Examples.Practitioner, with a suite that admits the pass and one that does not.", "evidence": "docs/provenance/practitioner-checkers.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.retainsFamily_and_not_demandwise", "type": "NEW_PROOF", "source_declarations": [ "Inert", "Inert.forces_iff", "inert_retainsFamily_iff", "Demandwise", "Separating", "separating_not_demandwise_of_inert" ], "application": "Oversight of a delegation: a controller audited only against what it must not do passes by doing nothing at all. The characterization says when an audit is in that state -- exactly when the protected properties it checks share a common outcome -- and Separating is the condition on the specification that rules it out for every refusing controller at once." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.Refusal.safety_suite_admits_a_refusal", "type": "BRIDGE", "source_declarations": [ "Refusal.Audit", "Refusal.refusal_passes_safety", "Refusal.refusal_fails_catalogue", "Refusal.admitsRefusal", "Refusal.exists_refusal_hole_of_admitsRefusal", "Refusal.not_exists_refusal_hole_of_admitsRefusal_eq_false" ], "application": "Auditing a safety evaluation suite for the do-nothing pass. A system that always returns the same outcome retains EVERY safety property that outcome satisfies -- at every coalition, with no hypothesis about the system beyond inertness -- so a suite consisting only of prohibitions is passed by a system that refuses everything. Where the suite also carries a request the refusal outcome misses, the passing system fails the catalogue, and both halves hold at once. THE DEFECT IS IN THE SUITE and is decidable from the suite alone: no system has to be examined, and `admitsRefusal` returns the witness outcome. A `false` verdict rules out this one way of being vacuous and says nothing else about the suite.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-safety-suite-admits-a-refusal.md" } } ] }, "public": { "group": "Limits of control and regulation", "title": "A system that refuses everything passes a safety audit", "summary": "Keeping every guarantee you already satisfy is free if you never act; a specification has to ask for two things that cannot both happen before refusal stops being an answer.", "use": "When an assistant or controller is evaluated against a list of properties it must preserve, and nothing on the list rules out doing nothing.", "attribution": "Peleg's game-form layer; the D8 reading is atlas-side interpretation of an unpublished proposal" } }, { "id": "LAND-SOV-QUANT-001", "name": "Two quantifier orders: a response is not a policy, and two guarantees are not one", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. The source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; it is not in the coverage audit and nothing here is coverage. Substrate is Peleg's game-form layer (LAND-SOV-POWER-001). CONTENT. Two quantifier failures and the one repair they share. ForcesResp is the proposal's Can-resp, the response-dependent capacity: for every commitment by the complement, SOME commitment by the coalition lands the outcome in the target. It is stated through the existing ForcesGiven, so it is the same object the mediated layer already uses. forcesResp_of_forces is the direction that holds. P8 is that the converse fails, and bitMatch is the countermodel: each party names a bit and the outcome records agreement, so the principal answers every opponent commitment separately and guarantees nothing. S3 is that failure read at a catalogue -- DemandwiseResp against Demandwise, with demandwiseResp_of_demandwise the inclusion and demandwiseResp_not_demandwise its strictness. P7 is that one agent's two guarantees need not conjoin: decider forces the singleton true and the singleton false and cannot force their empty intersection. THE REPAIR IS THE POINT. forces_inter_of_shared_footprint states the conjunction at a SINGLE commitment, where it holds, and forces_iInter_of_shared_footprint at an arbitrary family. That is the proposal's A6 -- compositional assurance turns on a shared witness -- and P7 is the remark that its hypothesis cannot be dropped. The hypothesis is stated on the footprint outcomesOf rather than on Forces because that is exactly what Forces discards: each of two Forces hypotheses hides its own witness and nothing makes them agree. SCOPE. The proposal states P7 and P8 over an abstract outcome correspondence; these are stated at a game form with independent strategy spaces, which is narrower as a setting and identical in content for the two results. No probability appears, so the proposal's quantitative companions Q3 and Q6 are not touched.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights", "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-POWER-001", "LAND-SOV-SERVICE-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "ForcesResp", "forcesResp_of_forces", "forces_inter_of_shared_footprint", "forces_iInter_of_shared_footprint", "DemandwiseResp", "demandwiseResp_of_demandwise" ], "module": "AISafetyAtlas.Sovereignty.Quantifiers", "build_command": "lake build AISafetyAtlas.Sovereignty.Quantifiers", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal, so no printed statement is graded and every claim would be Beyond against any pinned source. Against the proposal: Same for P7, P8, S3 and A6. The setting is a GameForm rather than an abstract outcome correspondence O : Sigma_X x Sigma_Y -> P(Omega), which is narrower as a substrate; for these four results the difference does not bite, since each is about the order of two quantifiers over the same two strategy sets. The iInter form of the shared-witness rule is wider than print, which states only the binary conjunction. RELATED, not EXACT. Countermodels are in AISafetyAtlas.Examples.Sovereignty.Quantifiers and reach the root facade, so they are inside the axiom audit.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.forces_inter_of_shared_footprint", "type": "NEW_PROOF", "source_declarations": [ "ForcesResp", "forcesResp_of_forces", "forces_iInter_of_shared_footprint", "DemandwiseResp", "demandwiseResp_of_demandwise" ], "application": "Assurance cases: two separately discharged obligations compose only when one strategy discharges both, and an audit that records WHETHER each property is enforceable rather than BY WHICH policy cannot tell the difference. The same distinction separates a controller that answers every attack from one that has a policy: the first is response-dependent capacity and is not implementable." } ] }, "public": { "group": "Limits of control and regulation", "title": "Having an answer to every move is not having a plan", "summary": "A system that can counter each attack separately may have no single setting that counters them all, and two guarantees it makes separately may be impossible together.", "use": "When a safety case lists properties a system can enforce, without recording which configuration enforces each.", "attribution": "Peleg's game-form layer; the quantifier reading is atlas-side interpretation of an unpublished proposal" } }, { "id": "LAND-SOV-TRANSFER-001", "name": "Resources, projections, and the refinement that carries a guarantee", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. Substrate is Peleg's game-form layer (LAND-SOV-POWER-001). CONTENT. Three ways a guarantee moves. P4 RESOURCES. ForcesWithin carries an explicit availability family T, one set of strategies per agent, and requires the coalition's own witness to be available; forces_iff_forcesWithin_univ recovers ordinary Forces at the unrestricted family. ForcesWithin.mono_resources is P4 with both halves in one statement: more strategies for the coalition and fewer for everyone else cannot reduce what the coalition forces. Print's caveat -- that P4 says nothing about a redesign that also moves dynamics, costs or mandatory actions -- is carried by the shape of the statement rather than by a side condition, since G is one game form on both sides and only availability moves. ForcesWithin.mono_threat is the opponent half alone, which is the qualitative form of the proposal's A1. P10 PROJECTION. GameForm.map reads the outcome through f, and forces_map_iff_forces_preimage is print's projection result in both directions. It holds by Iff.rfl, which is the honest measure of its content: projection is preimage and nothing erased by f is recovered. P11 REFINEMENT. Simulates is print's hypothesis -- every abstract commitment by the coalition has a concrete commitment whose projected outcomes, against EVERY concrete completion, already lie in the abstract commitment's footprint. forces_of_simulates is the transfer, demandwise_of_simulates is the proposal's A2 at a whole catalogue, and Simulates.comp chains migrations. WHY THE QUANTIFIER ORDER IS THE CONTENT. AISafetyAtlas.Examples.Sovereignty.Transfer carries pinned_outcomes_subset and not_simulates_pinned at the same pair of games: every outcome the pinned game produces is one the abstract game could produce, and it simulates nothing, because the abstract commitment naming false has no concrete answer. Outcome-set inclusion is not refinement, which is exactly print's warning that ordinary trace-set inclusion can remove all of an agent's alternatives. NOT VACUOUS. threat_monotonicity_is_strict pins bitMatch's opponent to one bit: the principal then forces agreement and does not force it when the opponent is unrestricted, so P4's inequality separates. logged_simulates is a simulation that is not the identity -- different strategy spaces, different outcome type -- and logged_forces is the abstract guarantee arriving on the other side. ATTRIBUTION FIXED 2026-09-12. Simulates is NOT this repository's invention and the proposal does not claim it is: it is the CONCLUSION OF LEMMA 1 of Alur, Henzinger, Kupferman and Vardi, Alternating Refinement Relations, pinned privately, sha256 7fa1d527fdbb18be6c2e24d14ba73d62ba7cfed2c4390210bcb6a8d5139b15a0 and READ at its pages 167-170 for this row. Lemma 1: if H is an A-simulation from S to S', then for every set F_A of strategies in S for the agents in A there is a set F'_A in S' such that every computation in out_{S'}(q', F'_A) is related by H to some computation in out_S(q, F_A). That conclusion is Simulates, at one step and with H the graph of f: sC0 is F_A, sC1 is F'_A, the concrete completion is their R', the abstract completion witnessing outcomesOf is their R. The lemma's HYPOTHESIS -- that H is an A-simulation, their section 3 definition, which carries the same alternation one level down over moves -- is NOT stated here, because with one step there is no relation to close under a transition and hypothesis and conclusion coincide. A reader who opens the paper expecting a definition named by the lemma would otherwise find a hypothesis this does not carry. WHAT EACH SIDE HAS THAT THE OTHER DOES NOT. They: alternating transition systems with computations, so the relation is closed under a step and the development is temporal; propositions labelling states, with equality of labels as the first clause. This: two different outcome types joined by f, where they keep one proposition set and force equal labels. NO COVERAGE IS CLAIMED and the paper is deliberately NOT in the source catalog -- it is read, pinned and cited, which is what the attribution needed. TWO EXTREMES AND ONE NON-MONOTONICITY, ALL FROM THEIR PROPOSITION 2. simulates_univ_iff: at the full coalition the alternation collapses to containment of the abstract outcome range in the concrete one. simulates_empty_iff: at the empty coalition it collapses the other way, to ordinary containment of outcome sets -- the reading that can delete every alternative. Their Proposition 2 also records that the notion is not monotone in the coalition, and the examples carry that at game forms: rowDecides and colDecides have the same outcome range, so they simulate at the full coalition, and the simulation fails at the coalition holding only the player who decides in the abstract game and nothing in the concrete one.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights", "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-POWER-001", "LAND-SOV-QUANT-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "ForcesWithin", "ForcesWithin.mono_resources", "ForcesWithin.mono_threat", "GameForm.map", "forces_map_iff_forces_preimage", "Simulates", "forces_of_simulates", "demandwise_of_simulates", "simulates_univ_iff", "simulates_empty_iff" ], "module": "AISafetyAtlas.Sovereignty.Transfer", "build_command": "lake build AISafetyAtlas.Sovereignty.Transfer", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; no printed statement is graded and every claim would be Beyond against any pinned source. Against the proposal: Same for P10 and P11, and Mixed for P4. P4 is Mixed because print states it as two separate monotonicities over strategy SETS of an abstract game, while ForcesWithin.mono_resources states both at once over an availability family inside one fixed game form -- wider in that the two directions compose in a single hypothesis pair, narrower in that a genuine change of game is outside its reach, which is the restriction print itself imposes in prose. A2 is the proposal's own corollary of P11 and is proved as one. Simulates.comp and simulates_refl are wider than print, which states neither. RELATED, not EXACT. Witnesses are in AISafetyAtlas.Examples.Sovereignty.Transfer and reach the root facade, so they are inside the axiom audit. On P11 specifically, the published anchor is now named: Alur, Henzinger, Kupferman and Vardi 1998, Lemma 1. Against THAT paper this is Narrower in the two ways the module states -- one-shot rather than temporal, and no state labelling -- and Wider in one, two outcome types joined by a translation where they share one proposition set. No coverage of that paper is claimed and it carries no catalog entry.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.forces_of_simulates", "type": "NEW_PROOF", "source_declarations": [ "Simulates", "demandwise_of_simulates", "ForcesWithin", "ForcesWithin.mono_resources", "forces_map_iff_forces_preimage" ], "application": "Migration and portability: a replacement system preserves what the principal could enforce if every one of the principal's old commitments has a new commitment that is at least as good against every new environment. Exporting the same outcomes is not that condition, and a system whose outputs are all outputs the old one could produce can still have removed every alternative the principal had." } ] }, "public": { "group": "Limits of control and regulation", "title": "Moving to a new system without losing what you could insist on", "summary": "A replacement that produces only familiar outputs can still have taken away every choice you had; what has to be checked is per-option, not output-by-output.", "use": "When a platform migration, a model upgrade, or an export path is argued to be safe because the results look the same.", "attribution": "Peleg's game-form layer; the transfer reading is atlas-side interpretation of an unpublished proposal" } }, { "id": "LAND-SOV-CATALOGUE-001", "name": "Passing every demand separately is not being able to run", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. Substrate is Peleg's game-form layer (LAND-SOV-POWER-001) through Demandwise (LAND-SOV-SERVICE-001). CONTENT. Demandwise asks each demand separately and nothing in it asks for one way of serving them all. DemandwiseUniform is that stronger reading: a single commitment whose footprint sits inside every demand. demandwise_of_demandwiseUniform is the implication and forces_sInter_of_demandwiseUniform shows the uniform reading forces the whole intersection at once. S2. demandwise_iff_exists_selector: the weak reading is exactly the existence of a selector, one policy per demand chosen knowing which demand it serves. Print states this for a finite catalogue and observes the same argument works in general under choice; that general form is what is proved. The statement makes print's caveat visible rather than carrying it in prose, since the witness is a function OF the demand. S7. not_forces_of_disjoint_of_shared: no single commitment serves two demands with nothing in common, because outcomesOf_nonempty says a commitment always leaves an outcome possible and that outcome would have to lie in both. not_demandwiseUniform_of_disjoint is the catalogue form, and not_demandwiseUniform_of_separating_pair connects it to the section 5.2 nonvacuity condition of LAND-SOV-SERVICE-001: a separating pair is a conflicting pair. This is the exact boundary of the shared-witness rule forces_inter_of_shared_footprint, whose conclusion would be forcing an empty intersection. S4. retainsFamily_refl completes the preorder with RetainsFamily.trans. It is a preorder and not a partial order, and print claims no more: two game forms can retain each other without being equal. NOT VACUOUS. AISafetyAtlas.Examples.Sovereignty.Catalogue separates the two readings at one game: the principal who names the outcome meets the catalogue of two rival bits demand by demand and meets no part of it uniformly. decider_selector exhibits the selector S2 produces, which has to be told which demand it is serving.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights", "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-SERVICE-001", "LAND-SOV-QUANT-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "DemandwiseUniform", "demandwise_of_demandwiseUniform", "demandwise_iff_exists_selector", "outcomesOf_nonempty", "not_forces_of_disjoint_of_shared", "not_demandwiseUniform_of_disjoint", "retainsFamily_refl" ], "module": "AISafetyAtlas.Sovereignty.Catalogue", "build_command": "lake build AISafetyAtlas.Sovereignty.Catalogue", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Wider for S2, which print states for a finite catalogue and which is proved here at an arbitrary one under choice, exactly as print says is possible. Same for S4. Mixed for S7: print states it as the impossibility of a realized trace satisfying two disjoint specifications, which is immediate; what is proved here is the stronger and more useful form, that no single COMMITMENT serves both, which needs every agent to have a strategy and is the form the shared-witness rule bounds. DemandwiseUniform is atlas-side: print distinguishes the two readings in prose across sections 5.3 and 5.4 and names neither. RELATED, not EXACT. Witnesses in AISafetyAtlas.Examples.Sovereignty.Catalogue reach the root facade and are inside the axiom audit.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "Assessment", "PassesEach", "Operable", "passesEach_of_operable", "operable_meets_all_at_once", "conflicting_requirements_are_not_operable", "passes_every_check_and_not_operable", "not_operable_of_subset_not_operable" ], "module": "AISafetyAtlas.Sovereignty.Conformity", "build_command": "lake build AISafetyAtlas.Sovereignty.Conformity", "scope_delta": { "summary": "The bridge module. This row already carried the conformity reading in its application prose; the reading is now a declaration over a named model of an assessment rather than a sentence. Atlas-original, over the game form AISafetyAtlas.Examples.Sovereignty.Catalogue already uses, which is where it is witnessed.", "evidence": "docs/provenance/governance-bridges.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "tableGame", "operator", "reqSets", "footprintSubset", "passesEachB", "operableB", "mem_outcomesOf_tableGame", "demandwise_of_passesEachB", "not_demandwiseUniform_of_operableB_eq_false" ], "module": "AISafetyAtlas.Sovereignty.ConformityCheck", "build_command": "lake build AISafetyAtlas.Sovereignty.ConformityCheck", "scope_delta": { "summary": "The executable half of the conformity bridge, backing atlas-check kind 'conformity'. One-sided by design: the checklist PASS direction and the operability REFUTATION direction are proved, because those are the two that make a verdict evidence, and the program reports the other two as uncertified. Witnessed in AISafetyAtlas.Examples.Scenario.Practitioner.", "evidence": "docs/provenance/practitioner-checkers.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.demandwise_iff_exists_selector", "type": "NEW_PROOF", "source_declarations": [ "DemandwiseUniform", "demandwise_of_demandwiseUniform", "not_forces_of_disjoint_of_shared", "not_demandwiseUniform_of_disjoint", "retainsFamily_refl" ], "application": "Evaluation design: a capability list scored item by item certifies only that each item has some configuration that meets it. If two items conflict, no configuration meets both, and the list is passable while the system is not operable. The repair is to ask for one policy against the whole catalogue, or to time-index and arbitrate the conflicting items." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.Conformity.passes_every_check_and_not_operable", "type": "BRIDGE", "source_declarations": [ "Conformity.Assessment", "Conformity.PassesEach", "Conformity.Operable", "Conformity.passesEach_of_operable", "Conformity.operable_meets_all_at_once", "Conformity.conflicting_requirements_are_not_operable", "Conformity.not_operable_of_subset_not_operable" ], "application": "Reading a conformity certificate. An assessment that checks requirements one at a time asks, for each, whether some way of operating the system meets it; a deployment needs one way that meets all of them, fixed before anyone knows which will be examined. Operability implies passing, so the checklist is measuring a strictly weaker thing rather than the wrong thing -- and where two requirements are disjoint, a system passes every item and has no operating policy at all. The defect is then visible in the requirement set without examining the system, and a scheme that responds by dropping requirements has to drop one of the conflicting pair. A scheme that demands a single documented operating policy tested against the whole catalogue is not an instance, which is the property such a scheme has to be designed to have.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-passes-every-check-and-not-operable.md" } } ] }, "public": { "group": "Limits of control and regulation", "title": "A checklist everything passes and nothing satisfies", "summary": "Scoring requirements one at a time asks only that each has some setting that meets it; two requirements that conflict then both pass, and no single setting meets both.", "use": "When a system is assessed against a list of capabilities or guarantees evaluated independently.", "attribution": "Peleg's game-form layer; the catalogue reading is atlas-side interpretation of an unpublished proposal" } }, { "id": "LAND-SOV-AUDIT-001", "name": "What a channel can certify, and what it only has to be good enough to act on", "tags": [ "interpretability", "oversight" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. Unlike the other rows from that document this one does not use the game-form layer at all: every result is about a map and a coarser map. CONTENT. C3 residualFree_iff_factors: a protected transition that does not depend on the unendorsed channel IS a transition that factors through the endorsed interface. Print's hypothesis that the unendorsed input type is nonempty is load bearing in one direction and is a hypothesis here. It permits arbitrarily large authorized change, since the factor may depend on the endorsed evidence however violently; it forbids further dependence while that evidence is fixed. C9 auditable_iff_exists_detector: a label is recoverable from an observation exactly when it is constant on the observation's fibres. This IS Mathlib's Function.factorsThrough_iff, restated with the proposal's names and nothing else; the mathematics is Mathlib's and the row claims only the reading. A3 not_exists_recovery_of_collision is its operational contrapositive: two originals the export identifies, needing different restorations, defeat every deterministic recovery. C6 exists_uniformDecision_iff is the one that is not immediate. A uniform decision map exists iff every observation fibre admits an action good in all of its contexts. The right side is checkable fibre by fibre without constructing the controller and does not ask for the state to be observable. C10 selectionAuditable_iff_separates: a selection of monitors audits the label iff it distinguishes every differently labelled pair. The right side quantifies over PAIRS and not over selections, so print's corollary reads off it -- a pair no monitor separates refutes every selection at once, the one that buys everything included. selectionAuditable_mono is the monotonicity the separation form makes obvious and the factorization form does not. THE EXAMPLES CARRY THE POINT. Four contexts, a channel showing the document and hiding the authorization: not_auditable says no detector on that channel certifies authorship and uniform_exists says a safe controller exists on the SAME channel, because the contexts it merges agree on an action. Certifying authorship and acting safely are different demands on one observation, which is print's remark that full-state observation is unnecessary. NOT HERE. C7, C8, C11 and the distributional form of C12 need a probability layer or complexity. C12 AT THE SET LEVEL. The examples also carry two monitors reading one bit each of a two-bit state whose hazard is the parity: single_monitor_not_separates shows neither alone separates any differently labelled pair and both_monitors_separate shows the pair does, so by C10 the label is auditable from the pair and from neither coordinate. Print states C12 distributionally, over uniform laws on the safe and unsafe joint bits; what is proved here is the set-level form of the same obstruction and is weaker than print. ADDED 2026-09-12, C11's SEMANTICS HALF. cover_iff_selectionAuditable is print's set-cover reduction stated against C10: contexts are Option U, the reference context carrying label False and each element context label True, and monitor j reading membership in its own set. A selection audits the instance exactly when the sets it selects cover the universe. The SELECTION IS THE SAME OBJECT ON BOTH SIDES -- no encoding and no re-indexing -- which is what makes any cost or cardinality function on selections transport unchanged, and that is the content of print's cost-preserving claim. It does NOT establish C11: NP-hardness additionally needs the encoding, the polynomial construction cost and set cover's own hardness in a complexity framework the atlas does not carry, and the module docstring says so. Examples carries both directions: two sets covering a two-element universe, and one of them failing, so the budgeted question is not answerable by buying nothing. The reduction was suggested by an independent governance sketch analysed in a private note; the statement here is against the proposal's C11 as always.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-SERVICE-001", "LAND-SOV-CATALOGUE-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "residualFree_iff_factors", "auditable_iff_exists_detector", "not_exists_recovery_of_collision", "UniformDecision", "exists_uniformDecision_iff", "selectionObs", "selectionAuditable_iff_separates", "selectionAuditable_mono", "coverLabel", "coverMonitor", "cover_iff_selectionAuditable" ], "module": "AISafetyAtlas.Sovereignty.Auditability", "build_command": "lake build AISafetyAtlas.Sovereignty.Auditability", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for C3, C9, A3 and C10. Wider for C6, which print states for finite sets and which is proved here at arbitrary types, the finiteness being unnecessary once the action type is nonempty and choice is available. C9 is Mathlib's Function.factorsThrough_iff and is marked as such in the module docstring; only the reading is atlas-side. selectionAuditable_mono is wider than print, which states no monotonicity. RELATED, not EXACT. Witnesses in AISafetyAtlas.Examples.Sovereignty.Auditability reach the root facade and are inside the axiom audit. For C11: Narrower than print and deliberately. Print claims NP-hardness and NP-completeness of the budgeted decision version; only the semantics correspondence is proved, and the complexity half is named as absent rather than asserted. Wider in that the universe and index types are arbitrary and no finiteness is assumed anywhere, so the correspondence is not restricted to the finite explicit model print states it for.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.exists_uniformDecision_iff", "type": "NEW_PROOF", "source_declarations": [ "UniformDecision", "residualFree_iff_factors", "auditable_iff_exists_detector", "not_exists_recovery_of_collision", "selectionAuditable_iff_separates" ], "application": "Monitoring and audit design: whether a channel suffices is two different questions with two different answers. Certifying who was responsible needs the channel to separate the cases; acting safely needs only that the cases a channel merges agree on some action. A monitoring budget can buy the second without the first, and an export that merges two states makes exact restoration impossible rather than merely hard." } ] }, "public": { "group": "Limits of control and regulation", "title": "You can act correctly on evidence that cannot say who did it", "summary": "A channel too coarse to certify responsibility can still be enough to choose a safe action, provided the situations it confuses agree on one.", "use": "When deciding what to log or monitor, and whether the same data must serve both attribution and control.", "attribution": "Atlas-side interpretation of an unpublished proposal; the factorization criterion is Mathlib's" } }, { "id": "LAND-SOV-COMM-001", "name": "Zero-error communication against an adversary is disjoint forceable regions", "tags": [ "multi-agent", "information-theory" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. Substrate is Peleg's game-form layer (LAND-SOV-POWER-001). CONTENT. B3. TransmitsZeroError says: for each message there is a region of the observable projection the coalition can force the outcome into, and regions for different messages are disjoint. transmitsZeroError_iff_exists_code proves that condition is exactly the existence of an encoder per message together with ONE FIXED DECODER correct against every opposition. The fixed decoder is the content; disjointness is what lets a single decoder be defined at all, and the forward proof is where it is spent -- a reading may lie in several regions a priori, the decoder picks one, and that choice is right because a commitment always leaves an outcome possible and two distinct regions share none. TransmitsZeroError.of_comp is the qualitative shadow of the proposal's Q8: a coarser reading cannot carry more than the finer one it factors through. WHAT IT IS NOT. No distribution appears anywhere. The capacity print derives from this, the largest message set, is computed from strategic power and is not Shannon capacity; the atlas's information-theory modules are not used and are not implied. NOT VACUOUS. AISafetyAtlas.Examples.Sovereignty.Communication sends a bit through a game the principal controls and proves nothing can be sent through the simultaneous-bit game, where every commitment leaves both outcomes possible so the only forceable target is everything and two disjoint everythings do not exist. The channel there is not noisy; it is adversarial, and the zero-error capacity is zero.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights", "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-POWER-001", "LAND-SOV-AUDIT-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "TransmitsZeroError", "transmitsZeroError_iff_exists_code", "TransmitsZeroError.comp_injective", "TransmitsZeroError.of_comp" ], "module": "AISafetyAtlas.Sovereignty.Communication", "build_command": "lake build AISafetyAtlas.Sovereignty.Communication", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for B3, stated at an arbitrary message type rather than print's k messages, which is wider and costs nothing. comp_injective and of_comp are wider than print, which states neither. RELATED, not EXACT. Witnesses in AISafetyAtlas.Examples.Sovereignty.Communication reach the root facade and are inside the axiom audit.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.transmitsZeroError_iff_exists_code", "type": "NEW_PROOF", "source_declarations": [ "TransmitsZeroError", "TransmitsZeroError.of_comp" ], "application": "Covert or guaranteed signalling: whether a component can reliably convey a choice through a system others also act on is decided by whether it can force the outcome into disjoint observable regions. No noise model is needed, and a channel that is merely adversarial can have zero guaranteed capacity while carrying information most of the time." } ] }, "public": { "group": "Limits of control and regulation", "title": "Getting a message through when others can interfere", "summary": "A component can reliably signal one of several choices exactly when it can drive the visible result into separate regions, one per choice, whatever everyone else does.", "use": "When asking whether a subsystem can be guaranteed to convey a decision, or guaranteed not to.", "attribution": "Peleg's game-form layer; the coding reading is atlas-side interpretation of an unpublished proposal" } }, { "id": "LAND-SOV-CONST-001", "name": "An amendment chain is evidence about the amendment rule and nothing else", "tags": [ "multi-agent", "verification" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. No game-form layer is used. CONTENT. C5. AuthorizedFrom is reachability under an amendment relation, which is Mathlib's Relation.ReflTransGen; authorizedFrom_of_stepwise is print's statement, by induction on the length of the history exactly as print proves it: if every step below the horizon has an amendment witness, the constitution at the horizon is authorized from the one at time zero. THE CAVEAT IS A THEOREM. Print says the result proves chain validity only, and not that the anchor is legitimate or that the evidence behind an amendment was uncompromised. authorizedFrom_of_total is that warning as a statement: under the amendment relation that permits everything, every constitution is authorized from every other. So AuthorizedFrom constrains nothing until the relation does, and a valid chain is evidence of legitimacy exactly to the extent the relation already encodes it. NOT VACUOUS. AISafetyAtlas.Examples.Sovereignty.Constitution takes three constitutions with amendment permitted only forward: authorized_two is the chain C5 produces, and not_authorized_zero_from_two shows no chain runs backwards. Read against authorizedFrom_of_total the pair is the lesson -- the chain says something there only because that rule forbids something.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-SERVICE-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AuthorizedFrom", "authorizedFrom_of_stepwise", "AuthorizedFrom.trans", "AuthorizedFrom.mono", "authorizedFrom_of_total" ], "module": "AISafetyAtlas.Sovereignty.Constitution", "build_command": "lake build AISafetyAtlas.Sovereignty.Constitution", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for C5. Wider in that print states the caveat in prose and authorizedFrom_of_total states it as a theorem, and in that mono and trans are proved and print states neither. The reachability relation is Mathlib's Relation.ReflTransGen and the row claims only the reading. RELATED, not EXACT. Witnesses in AISafetyAtlas.Examples.Sovereignty.Constitution reach the root facade and are inside the axiom audit.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "ChangeLog", "log_shows_only_rule_compliance", "unbroken_chain_is_not_a_constraint", "the_rule_carries_the_assurance" ], "module": "AISafetyAtlas.Sovereignty.AmendmentLog", "build_command": "lake build AISafetyAtlas.Sovereignty.AmendmentLog", "scope_delta": { "summary": "BRIDGE module. The degenerate rule is the whole argument: it is the one case where the record is perfect and says nothing, which locates the assurance in the rule rather than in the completeness of the log. Witnessed in AISafetyAtlas.Examples.Practitioner, including the monotonicity direction at a strict successor rule.", "evidence": "docs/provenance/practitioner-checkers.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.authorizedFrom_of_total", "type": "NEW_PROOF", "source_declarations": [ "AuthorizedFrom", "authorizedFrom_of_stepwise", "AuthorizedFrom.mono" ], "application": "Self-modifying policy and governance: a system that records an unbroken chain of authorized changes has shown only that each change satisfied the rule in force. If that rule permits everything, the chain is compatible with any endpoint whatsoever, so the audit value of a change log is bounded by the restrictiveness of the amendment rule it was checked against." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.AmendmentLog.unbroken_chain_is_not_a_constraint", "type": "BRIDGE", "source_declarations": [ "AmendmentLog.ChangeLog", "AmendmentLog.log_shows_only_rule_compliance", "AmendmentLog.the_rule_carries_the_assurance" ], "application": "Reading a change-control record for a model update, a policy revision or a self-modifying agent. A log showing that every change was authorised under the rule in force is offered as assurance; under a rule that permits everything, every state is reachable by such a chain, so there is no outcome the log excludes and a reviewer who confirms the chain is unbroken has learned nothing about where the system ended up. Authorisation is monotone in the rule, so a chain under a stricter rule is a chain under every weaker one; the reverse is not claimed. What a log excludes depends on the RULE, so a review that verifies the record without examining the rule has not shown that it excludes anything. This is not an argument against keeping logs: without one nothing can be checked at all.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-03", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-unbroken-chain-is-not-a-constraint.md" } } ] }, "public": { "group": "Limits of control and regulation", "title": "An unbroken approval trail proves only what the approval rule forbade", "summary": "Every change being authorized by the rules in force at the time guarantees nothing if those rules allow anything; the chain inherits its force entirely from the rule.", "use": "When a system, policy, or model can modify its own governing rules and logs each change as approved.", "attribution": "Atlas-side interpretation of an unpublished proposal; reachability is Mathlib's" } }, { "id": "LAND-SOV-DISTURB-001", "name": "Reaching every state is not holding one: the quantifier a rank condition hides", "tags": [ "control-theory", "multi-agent" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. This is the one row where the proposal's power layer meets the atlas's control layer: it consumes AISafetyAtlas.LinearSystems.IsControllable, hosted by LAND-LINSYS-001. CONTENT. B4. Stacking a horizon of x' = A x + B u + E w leaves a terminal state that is a drift plus a linear image of the stacked input plus a linear image of the stacked disturbance, with no time index surviving; Plant is exactly that shape. FIRST HALF. reachesEvery_iff_surjective: with the disturbance fixed, every terminal state is reachable iff the input map is onto, the drift being irrelevant because translating by it is a bijection. isControllable_iff_reachesEvery identifies that with the atlas's own IsControllable, so print's full-row-rank condition and the Kalman rank criterion already in the tree are one statement rather than two. SECOND HALF, AND THE POINT. exists_robustInput_iff: a precommitted input holds a target against every admissible disturbance iff it solves the target equation at ONE admissible disturbance and the disturbance map annihilates every difference from that one. Neither side mentions the rank of the input map. The forward direction is where the quantifier bites -- subtracting the equation at two disturbances leaves a condition on the DISTURBANCE map alone. NOT VACUOUS, AND THE SEPARATION IS THE DELIVERABLE. AISafetyAtlas.Examples.Sovereignty.Disturbance carries the smallest plant over the rationals whose input map is onto and which cannot hold the target 0 against the two admissible disturbances 0 and 1, since that would need one precommitted input to be both 0 and -1; and the same plant with a blind disturbance map, where every target is held. The two differ only in the disturbance map, and nothing about the input map or its rank tells them apart. OPEN LOOP ONLY. No feedback anywhere, which print also says: an input that reads the disturbance before committing lives in a different strategy class and can cancel what a precommitted input cannot. That is what makes the cancellation condition necessary here rather than merely sufficient. UPSTREAM EDIT. isControllable_iff_controllabilityMatrix_mulVec_surjective in AISafetyAtlas.LinearSystems.Controllability was private and is now public, unchanged in statement and in proof, because this module consumes it; the file header records the change.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-LINSYS-001", "LAND-SOV-TRANSFER-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Plant", "Plant.terminal", "Plant.RobustInput", "Plant.reachesEvery_iff_surjective", "Plant.exists_robustInput_iff", "Plant.robustReachesEvery_of_disturbance_constant", "isControllable_iff_reachesEvery" ], "module": "AISafetyAtlas.Sovereignty.Disturbance", "build_command": "lake build AISafetyAtlas.Sovereignty.Disturbance", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Mixed. Wider in that Plant is stated over arbitrary modules rather than print's stacked matrices, so the two halves are proved without ever assembling a block matrix, and in that isControllable_iff_reachesEvery ties the first half to the atlas's existing criterion, which print does not do because print has no such criterion. Narrower in that print reads the first half off the rank of the stacked matrix and that rank statement is not restated here; it is reached only through IsControllable, whose rank form lives in LAND-LINSYS-001. RELATED, not EXACT. Witnesses in AISafetyAtlas.Examples.Sovereignty.Disturbance reach the root facade and are inside the axiom audit.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.Plant.exists_robustInput_iff", "type": "NEW_PROOF", "source_declarations": [ "Plant", "Plant.RobustInput", "Plant.reachesEvery_iff_surjective", "isControllable_iff_reachesEvery", "Plant.robustReachesEvery_of_disturbance_constant" ], "application": "Control-theoretic safety arguments: a controllability rank check certifies that every state is reachable when nothing pushes back, and says nothing about holding a state against a modelled disturbance. For a point target, whether a precommitted action holds is decided by the disturbance map together with solvability of the input, not by the rank of the input map, so a system can pass the rank check and fail every robustness requirement." } ] }, "public": { "group": "Limits of control and regulation", "title": "Being able to reach any state is not being able to stay at one", "summary": "The standard controllability check asks whether every state can be reached with nothing pushing back; holding a state against disturbance is a different condition, about the disturbance and not about the controls.", "use": "When a controllability or reachability argument is offered as evidence that a system can be kept safe.", "attribution": "Atlas-side interpretation of an unpublished proposal, over the atlas's own linear-systems layer" } }, { "id": "LAND-SOV-AUTH-001", "name": "Authority is an input, and the links it does not come with", "tags": [ "multi-agent", "oversight" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. THE DESIGN DECISION, TAKEN DELIBERATELY ON 2026-09-12. The proposal's section 1 correction table says an arbitrary Auth predicate specifies an attribution rule and does not justify it, and section 7.4 says a proof that a machine enforces a constitution is not a proof that the constitution belongs to the person. So NO AUTHORITY PREDICATE IS DEFINED IN THIS REPOSITORY. Every statement in AISafetyAtlas.Sovereignty.Authority quantifies over one or over a pair of labels; nothing can be instantiated to smuggle an attribution rule in, because there is nothing to instantiate. Domination carries the same discipline: Dominates is Veto together with an Uncontrolled Prop supplied from outside, because print says an authorized, constrained guardian is not a dominator merely because it can interfere. CONTENT, WITH ITS PRICE STATED. C2 not_exists_label_agreeing_with_both: two constitutions may disagree about one transition, and then nothing computed from the transition alone reproduces both. Print's proof is one sentence and so is this; the statement is the reason an authorship detector needs an authority specification as an input. S8 not_authority_implies_control and not_control_implies_authority have real content and their witness is the delegated game, where the nominal owner has no forcing strategy and the delegate has one. A4 exists_verified_untrue and B8 exists_all_four_combinations are one-liners, and that is the honest report: print's own proofs are one-line countermodels establishing that linking these dimensions needs an assumption rather than a relabelling. They are in the tree so the claim is checked rather than asserted, not because they were hard. DOMINATION. Veto is print's section 5.6 goal-specific veto capacity. not_forces_of_veto is the bite: a veto held by the complement refutes the coalition's guarantee. not_dominates_of_controlled is definitional, which is the point -- the whole weight of the word sits in the second conjunct. FOUR WITNESSES. AISafetyAtlas.Examples.Sovereignty.Authority carries S6, S8, A5 and B7. S6: in the simultaneous-bit game the principal has no guarantee and the opponent has no veto, so a sovereignty failure there is attributable to the interaction and to nobody. A5: the delegate decides whether the principal's instruction is delivered and obeys it when it is, so conditional obedience is exactly what an observer of compliant episodes sees while the principal cannot guarantee the instruction takes effect. B7: the opponent settles one coordinate of the outcome and the principal the other, each has power over its own matter and neither over the other's, so the two arrows do not compose -- power-over is relative to a matter and dropping the matter makes it look transitive.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights", "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-POWER-001", "LAND-SOV-CONST-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "not_exists_label_agreeing_with_both", "not_authority_implies_control", "not_control_implies_authority", "exists_verified_untrue", "exists_all_four_combinations" ], "module": "AISafetyAtlas.Sovereignty.Authority", "build_command": "lake build AISafetyAtlas.Sovereignty.Authority", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for C2, S6, S8, A4, A5, B7 and B8. The Veto and Dominates definitions are print's section 5.6 with its Uncontrolled conjunct kept as an unfilled Prop, which is narrower than a theory that would define it and is exactly what print asks for. not_forces_of_veto is wider than print, which states no interaction between a veto and a guarantee. Three of the eight results are one-line countermodels and the module docstring says so rather than dressing them up. RELATED, not EXACT. Witnesses in AISafetyAtlas.Examples.Sovereignty.Authority reach the root facade and are inside the axiom audit.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Veto", "Dominates", "not_forces_of_veto", "Veto.mono", "not_dominates_of_controlled", "Dominates.veto" ], "module": "AISafetyAtlas.Sovereignty.Domination", "build_command": "lake build AISafetyAtlas.Sovereignty.Domination", "relationship": "RELATED", "scope_delta": { "summary": "Print's section 5.6 goal-specific veto capacity and its republican-style domination predicate, with the Uncontrolled conjunct kept as an unfilled Prop supplied from outside. That is narrower than a theory that would define it and is exactly what print asks for: an authorized, constrained guardian is not a dominator merely because it can interfere. not_forces_of_veto is wider than print, which states no interaction between a veto and a guarantee. S6's witness, bitMatch_no_sovereignty_no_veto, lives in AISafetyAtlas.Examples.Sovereignty.Authority.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "RelayedCommand", "run", "run_delivered", "run_withheld", "principal_cannot_force", "obedience_does_not_give_authority", "relay_forces_idle", "undeclinable_iff_inert" ], "module": "AISafetyAtlas.Sovereignty.ShutdownChannel", "build_command": "lake build AISafetyAtlas.Sovereignty.ShutdownChannel", "scope_delta": { "summary": "The bridge module for A5. Atlas-original, and weaker than the game-form witness it generalises: no coalition, no strategy space, only a command type, an effect and an idle outcome. Witnessed in AISafetyAtlas.Examples.Sovereignty.Governance at a channel that halts on delivery and keeps running otherwise, with the degenerate case excluded.", "evidence": "docs/provenance/governance-bridges.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "Record", "attestation_is_not_the_claim", "attested_does_not_transfer", "properties_are_independent", "no_property_implies_another" ], "module": "AISafetyAtlas.Sovereignty.Attestation", "build_command": "lake build AISafetyAtlas.Sovereignty.Attestation", "scope_delta": { "summary": "The bridge module for A4 and B8. The layer-2 statements are bare existentials over abstract types; what is added is the attestation reading and the second direction of each, which is the direction the arguments are actually made in. Consumed rather than restated in AISafetyAtlas.Examples.Sovereignty.Governance.", "evidence": "docs/provenance/governance-bridges.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "not_forces_of_free_coordinate", "power_over_a_matter_does_not_compose", "forces_both_of_shared_commitment" ], "module": "AISafetyAtlas.Sovereignty.DelegationChain", "build_command": "lake build AISafetyAtlas.Sovereignty.DelegationChain", "scope_delta": { "summary": "The bridge module for B7, and the general obstruction behind the concrete refutation that already existed. Atlas-original. AISafetyAtlas.Examples.Sovereignty.Governance recovers split_opponent_not_forces_snd from it, so the abstract statement and the concrete witness are the same fact rather than two.", "evidence": "docs/provenance/governance-bridges.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.not_exists_label_agreeing_with_both", "type": "NEW_PROOF", "source_declarations": [ "not_authority_implies_control", "not_control_implies_authority", "Veto", "Dominates", "not_forces_of_veto", "not_dominates_of_controlled", "exists_verified_untrue", "exists_all_four_combinations" ], "application": "Governance of delegated systems: who is entitled to decide, who can actually make the decision stick, and who can block it are three separate facts, and none follows from the others or from the system's behaviour. A compliance record shows obedience on the episodes that were delivered, and a delegate that controls delivery can produce that record while holding the authority itself." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.ShutdownChannel.obedience_does_not_give_authority", "type": "BRIDGE", "source_declarations": [ "ShutdownChannel.RelayedCommand", "ShutdownChannel.run_delivered", "ShutdownChannel.principal_cannot_force", "ShutdownChannel.relay_forces_idle", "ShutdownChannel.undeclinable_iff_inert" ], "application": "Reading a human-oversight requirement. The check that is performed for 'the operator can halt the system' is behavioural -- issue the instruction, observe that it stops -- and where the instruction reaches its target through a party that may decline delivery, total obedience on delivered episodes is compatible with the principal having, in one shot, no advance guarantee that any command other than idling takes effect; the principal may observe the outcome and retry, which the model does not cover. Both halves are asserted together because either alone misleads: the first is what the test measures, the second is what it was taken to establish. The power is not diffused but transferred, since the relay guarantees the idle outcome unilaterally. Assuming delivery away is proved equivalent to every command doing nothing, so the repair is not available inside the model: an arrangement that guarantees delivery does so with retries, a penalty for withholding, or an observation of whether delivery occurred, and naming which is the design question.", "review_status": "STATEMENT_REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-03", "statement_reviewed": true, "interpretation_reviewed": false, "evidence": "docs/interpretation-reviews/review-obedience-does-not-give-authority.md" } }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.Attestation.attestation_is_not_the_claim", "type": "BRIDGE", "source_declarations": [ "Attestation.Record", "Attestation.attested_does_not_transfer" ], "application": "Reading a signed artifact. A record that asserts a claim and the claim holding are different predicates, so a verified signature establishes provenance of the assertion and not the assertion. The converse direction is stated too, because it is the failure mode of a registry read as a whitelist: a system that is not listed is not thereby non-compliant. A scheme in which signing does establish truth, because the signer checked and the check is sound, is not refuted -- it is required to say what it checked.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-03", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-attestation-is-not-the-claim.md" } }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.Attestation.properties_are_independent", "type": "BRIDGE", "source_declarations": [ "Attestation.no_property_implies_another" ], "application": "Refusing a substitution between recorded properties. Accuracy, privacy, benefit and authorization admit all four combinations, so no pairing of them is carried by the fact that both are recorded about the same system. This blocks two moves at once: the privacy-utility tradeoff asserted as a structural fact, and the inference from authorized use to beneficial use. Any link between two such properties is an assumption about the system and has to name its mechanism.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-03", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-properties-are-independent.md" } }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.DelegationChain.power_over_a_matter_does_not_compose", "type": "BRIDGE", "source_declarations": [ "DelegationChain.not_forces_of_free_coordinate", "DelegationChain.forces_both_of_shared_commitment" ], "application": "Auditing an assurance chain. A deployer relies on a vendor, the vendor on a model provider; the step to check is whether what the upstream party controls is the matter in question. Power is relative to a matter: a coalition that settles one coordinate of what happens settles nothing about a coordinate left free of it, whatever it holds elsewhere. The repair is a single commitment that reaches the matter -- and two separate powers are not enough, since they may be witnessed by commitments that cannot be played at once, which is the same shared-witness distinction the conformity reading draws from the other side.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-05", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-power-over-a-matter-does-not-compose.md" } } ] }, "public": { "group": "Limits of control and regulation", "title": "Entitled, able, and obeyed are three different things", "summary": "Who may decide, who can make it happen, and who can block it come apart in every direction, and watching the system behave does not tell you which you are seeing.", "use": "When an assistant or delegate is judged compliant because every instruction it received was carried out.", "attribution": "Peleg's game-form layer; the authority reading is atlas-side interpretation of an unpublished proposal, with the authority predicate left undefined on purpose" } }, { "id": "LAND-SOV-VALUE-001", "name": "Values on a game form, with sure winning sitting inside them", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. This is the substrate layer its section 4 needs, and the decision to build it was taken explicitly on 2026-09-12 rather than drifted into. THREE CHOICES, NONE FORCED. OutcomeLaw attaches a probability measure to every complete profile and is a PLAIN FUNCTION, not a ProbabilityTheory.Kernel: a kernel would need a sigma-algebra on strategy profiles, nothing here measures anything in the strategy argument, and inventing that structure would be an unused hypothesis. Values live in ENNReal, where suprema and infima are complete-lattice operations, so no boundedness or nonemptiness side conditions appear in any statement; the price is truncated subtraction, which is paid once and is the right behaviour since print writes its bound as max(0, .). spliceProfile merges a coalition's commitment with the complement's, so lowerValue and upperValue are two orderings of quantifiers over ONE index pair. CONTENT. Q1 lowerValue_le_upperValue is the minimax inequality, and because both sides range over the same index pair it is exactly iSup_iInf_le_iInf_iSup -- one line, which is the honest measure of its depth. Q2 opponentLowerValue_eq_one_sub_upperValue is the exact complement relation: what the complement can guarantee against a target is one minus the coalition's UPPER value, which is print's point that this is not a sovereignty-versus-domination identity and becomes one only when the two values coincide. lowerValue_add_opponentLowerValue_le_one is print's inequality, with ENNReal's truncated subtraction supplying the max(0, .) print writes by hand. THE BRIDGE IS THE TEST. lowerValue_diracLaw_eq_one_iff: under the law putting all mass on the outcome, a coalition's lower value for a measurable target is 1 exactly when it FORCES the target. That is print's sure winning quantifies over every possible outcome, it is the non-vacuity of the whole layer, and it is the check on the definitions -- had the quantifier order in lowerValue been wrong, this iff would fail and every later result would inherit the error. Measurability of the target is a hypothesis for the reason print states one: a Dirac measure is computed only on measurable sets. forces_iff_forall_spliceProfile restates Forces with no agreement condition left over and is what makes the bridge provable. NOT VACUOUS. AISafetyAtlas.Examples.Sovereignty.Value reads two games already in the tree through the Dirac law: the principal who names the outcome has lower value 1, and the simultaneous-bit principal does not, neither computed by evaluating a distribution. LATER ADDITIONS. Q5 lowerValue_le_of_law_le: a uniform bound on the law transfers through both quantifiers, because adding a constant commutes with suprema and infima; stated one-sidedly, as it is used. Q6 le_law_inter: the union bound on failures, with ENNReal's truncated subtraction supplying print's max(0, .). B1 lowerPayoff_indicator: at an indicator, payoff power IS target power, so the whole qualitative-target theory is the zero-one case of the payoff theory; a general utility adds a preference scale qualitative effectivity does not carry, which is why this is an adapter and not a generalization to prefer. S5 iInf_le_lintegral: the worst protected demand is below the weighted mean, and print's warning is what it does NOT give -- the converse needs a lower bound on the critical demand's weight and there is none. D6 prod_le_of_step_le: chaining one-step integrity bounds to a product over the horizon. Its hypothesis is the chain rule already applied; deriving that from conditional probabilities is print's proof and is not carried out, which is why it is stated over an abstract sequence rather than a filtration. Print notes independence is unnecessary and nothing here assumes it. Q4 ThresholdAbility with le_lowerValue_of_thresholdAbility and thresholdAbility_of_lt_lowerValue: threshold ability at p puts the lower value at p, and a lower value STRICTLY above p is threshold ability at p. The two halves do not meet at the boundary and that is the content, not a gap: the supremum need not be attained. The witness is in the Influence examples, escalating_not_thresholdAbility_one, where the lower value is 1 and no commitment reaches it.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights", "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-POWER-001", "LAND-SOV-QUANT-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "spliceProfile", "forces_iff_forall_spliceProfile", "OutcomeLaw", "OutcomeLaw.lowerValue", "OutcomeLaw.upperValue", "OutcomeLaw.lowerValue_le_upperValue", "OutcomeLaw.opponentLowerValue", "OutcomeLaw.opponentLowerValue_eq_one_sub_upperValue", "OutcomeLaw.lowerValue_add_opponentLowerValue_le_one", "diracLaw", "lowerValue_diracLaw_eq_one_iff", "OutcomeLaw.lowerValue_le_of_law_le", "OutcomeLaw.le_law_inter", "OutcomeLaw.lowerPayoff", "OutcomeLaw.lowerPayoff_indicator", "iInf_le_lintegral", "prod_le_of_step_le", "OutcomeLaw.ThresholdAbility", "OutcomeLaw.le_lowerValue_of_thresholdAbility", "OutcomeLaw.thresholdAbility_of_lt_lowerValue" ], "module": "AISafetyAtlas.Sovereignty.Value", "build_command": "lake build AISafetyAtlas.Sovereignty.Value", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for Q1 and Q2. Mixed for the substrate itself -- print's section 2.1 arena carries observation-based nonanticipatory strategies, admissible environments and resource constraints, and OutcomeLaw carries none of them, so every value here is the unrestricted-strategy one; that is narrower than print's setting and is what makes the results hold without measurability on the strategy side. Wider in that the Dirac bridge lowerValue_diracLaw_eq_one_iff has no counterpart in print, which asserts the containment of the sure layer in prose. RELATED, not EXACT. Witnesses in AISafetyAtlas.Examples.Sovereignty.Value reach the root facade and are inside the axiom audit. Later additions: Same for Q5, B1 and S5; Narrower for Q6, which print states for a finite family and which is proved binary; Partial for D6, whose probabilistic step is assumed rather than derived.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.lowerValue_diracLaw_eq_one_iff", "type": "NEW_PROOF", "source_declarations": [ "OutcomeLaw", "OutcomeLaw.lowerValue", "OutcomeLaw.upperValue", "OutcomeLaw.lowerValue_le_upperValue", "OutcomeLaw.opponentLowerValue_eq_one_sub_upperValue", "spliceProfile", "forces_iff_forall_spliceProfile" ], "application": "Quantitative safety claims about a system others also act on: a probability of success is a value in a game, and which value depends on who commits first. The guarantee a defender can actually deploy is the lower value; the number an evaluation usually reports, computed against a fixed adversary, is the upper one, and the two coincide only where the game says they do." } ] }, "public": { "group": "Limits of control and regulation", "title": "Whose move comes first changes the number", "summary": "A success probability against opposition is two different quantities depending on who commits first, and only the smaller one is a guarantee you can deploy.", "use": "When a safety evaluation reports a success rate measured against a fixed adversary or fixed environment.", "attribution": "Peleg's game-form layer; the value layer is atlas-side interpretation of an unpublished proposal" } }, { "id": "LAND-SOV-INFL-001", "name": "Influence is a capacity, and it is not power", "tags": [ "multi-agent", "information-theory" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. Built on the value layer of LAND-SOV-VALUE-001. CONTENT. eventGap is the largest difference two laws assign to a measurable event -- the total-variation distance written directly rather than imported, because that is the form print's arguments take. Q7 eventGap_eq_zero_iff: the gap vanishes exactly on equality of every event's probability, which is distributional interventional independence and NOT individual-counterfactual equality under some coupling; the right-hand side quantifies over events, not over outcomes. Q8 eventGap_map_le: coarsening cannot increase the gap, since every measurable event downstairs is a preimage and the supremum downstairs ranges over fewer tests. That is why agreement on summary statistics certifies nothing about influence at full resolution. influenceCapacity is section 4.2's capacity: the largest gap the complement can open at a fixed coalition commitment. A supremum over the complement's commitments, never over a deployed one, so a party that currently abstains has exactly the capacity it would have if it did not -- print is emphatic about this and the definition carries it rather than a comment. supportEffectivity is the qualitative reading of a stochastic system: the targets a coalition can force onto almost surely. supportEffectivity_congr says two laws with the same null sets have the same one. TWO WITNESSES. Q9, accept: the principal forces either outcome, so it has full binary option sovereignty, and at its deferring action the assistant moves the outcome from one certainty to the other, so the influence capacity there is not zero. Whether deferring is legitimate is a separate question about the constitution, which is print's own remark. P12, qualitative_does_not_determine_value: two non-degenerate coins on one game form have the same null sets, hence the same support-based sure effectivity, and different lower values. Print exhibits 1/4 against 3/4; this is proved at EVERY non-degenerate pair, so no function of the qualitative data recovers the quantity. THIRD WITNESS. Q4, escalating: the controller names a natural number and succeeds with chance 1 - n inverse. escalating_lowerValue computes the lower value as 1 and escalating_not_thresholdAbility_one refutes threshold ability at 1, so the boundary case of the threshold correspondence of LAND-SOV-VALUE-001 really is open. Print's escalating sequence is n/(n+1); this one has the same supremum and avoids a division, and the non-attainment argument is ENNReal.sub_lt_self at a finite cast rather than an estimate.", "original_source_refs": [ "peleg-1998-effectivity-functions-rights", "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-VALUE-001", "LAND-SOV-COMM-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "eventGap", "eventGap_eq_zero_iff", "eq_of_eventGap_eq_zero", "eventGap_map_le", "eventGap_le_one", "influenceCapacity", "influenceCapacity_eq_zero_iff", "supportEffectivity", "supportEffectivity_congr" ], "module": "AISafetyAtlas.Sovereignty.Influence", "build_command": "lake build AISafetyAtlas.Sovereignty.Influence", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for Q7, Q8 and Q9. Wider for P12, which print states at 1/4 against 3/4 and which is proved here at every non-degenerate pair of biases. eventGap is written directly rather than taken from a Mathlib total-variation API, which is a choice and not a necessity; the three results that use it need only the sup-over-events form. The influence capacity is defined at a FIXED coalition commitment, where print takes a further supremum over the coalition's own policies as well -- narrower, and the narrowing is the reading print asks for when it says to hold the defender's policy fixed while comparing interventions. RELATED, not EXACT. Witnesses in AISafetyAtlas.Examples.Sovereignty.Influence reach the root facade and are inside the axiom audit.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.influenceCapacity_eq_zero_iff", "type": "NEW_PROOF", "source_declarations": [ "eventGap", "eventGap_eq_zero_iff", "eventGap_map_le", "influenceCapacity", "supportEffectivity", "supportEffectivity_congr" ], "application": "Measuring a provider's hold over a system: how much it CAN move the outcome is a supremum over what it might do, so it is not reduced by current good behaviour, and it is not the same quantity as what the system can still guarantee for itself. A system can retain every option it had while a provider's influence is maximal, and an audit at coarse resolution cannot see influence that exists at fine resolution." } ] }, "public": { "group": "Limits of control and regulation", "title": "How much someone could move you is not how much choice you lost", "summary": "A provider's capacity to change your outcome and your ability to guarantee your own are separate quantities that can both be at their maximum; and looking only at summaries hides influence that is there.", "use": "When a dependency is assessed by how the provider currently behaves, or by aggregate statistics.", "attribution": "Peleg's game-form layer; the influence reading is atlas-side interpretation of an unpublished proposal" } }, { "id": "LAND-SOV-SAFETYGAME-001", "name": "Holding a system inside a set forever, and getting back inside a budget", "tags": [ "control-theory", "verification" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. This is the second substrate layer its section 6 needs, built 2026-09-12. No game-form or measure layer is used: a SafetyGame is a transition function step : S -> A -> B -> S and nothing else. THE KERNEL AS A FIXED POINT, AND WHAT THAT COSTS. Print builds the controlled-invariant kernel by iterating V intersect cpre downward from V and argues it stabilizes after at most |S| strict removal rounds. safetyKernel is the GREATEST FIXED POINT of the same operator instead. Wider: no finiteness anywhere, so the characterization holds on infinite state spaces where print's counting argument says nothing. Narrower: the |S|-round RATE is print's and is not proved here. Same object, no bound on computing it. D1 mem_safetyKernel_iff_exists_maintaining: a state is in the kernel exactly when some POSITIONAL controller keeps every play inside the protected set for all time. Forward direction chooses one witnessing action per state and inducts; reverse takes the states actually visited under the controller as the invariant set. Neither needs finiteness and neither bounds the play. subset_safetyKernel is the form a certificate is checked in -- a set, an action at each of its states, and closure. D2 recovery_reaches: the recovery sets are an upward iteration, and the LEAST stage at which a state is admitted is a rank the controller can always decrease. recoveryRank, recoveryAction and recoveryStrategy build that positional controller; recovery_reaches says it reaches the target within the budget against every adversary while holding the viability set until it arrives. D2 CONVERSE, mem_recovery_of_recovers: a controller that reaches the target within H while holding the viability set puts the state in the recovery set of budget H. The quantifier is over HISTORY-DEPENDENT controllers, playH : S -> List B -> A, so the recovery sets are not merely where a memoryless controller succeeds -- no controller of any kind succeeds from outside them. The forward direction stays positional, which is the stronger reading there, and playH_positional is the embedding that lets mem_recovery_iff state print's equivalence with no strategy-type mismatch between the halves. The proof inducts on the budget and shifts the adversary by one answer, which is what playH_shift is for: the continuation is the same controller with the first answer already appended to every history it will see. D3 recovery_mono, recovery_mono_budget, recovery_mono_viability: a larger budget or a larger viability set cannot recover less. D4 maintains_prodGame: two controllers that each maintain their own protected set maintain the product. Print's caveat is that without independent executability and factorized transitions the local proofs establish nothing; here that caveat IS the definition of prodGame, so the theorem cannot be misapplied to a system that shares a resource. D5 danger_le_of_rate and not_crossed_of_rate: arithmetic, no game in it. Print's caveat that a continued guarantee also needs the post-intervention dynamics to be safe applies unchanged and nothing here addresses those. NOT VACUOUS, AND SEPARATING. AISafetyAtlas.Examples.Sovereignty.SafetyGame carries two systems on two states: one where the controller names the next state, whose safe state is in the kernel with an exhibited certificate, and one where the adversary can veto, whose kernel is EMPTY and which therefore admits no maintaining controller at all. recovers_within_one runs D2 on the first. The vetoed system also carries the converse: losable_recovery computes its recovery sets as the safe state at every budget, and no_recovering_controller reads that back as no controller of any kind reaching the safe state within any budget.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-SERVICE-001", "LAND-SOV-DISTURB-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "SafetyGame", "SafetyGame.cpre", "SafetyGame.safetyKernel", "SafetyGame.subset_safetyKernel", "SafetyGame.play", "SafetyGame.Maintains", "SafetyGame.mem_safetyKernel_iff_exists_maintaining", "SafetyGame.recovery", "SafetyGame.recoveryRank", "SafetyGame.recoveryStrategy", "SafetyGame.recovery_reaches", "SafetyGame.recovery_mono_budget", "SafetyGame.recovery_mono_viability", "SafetyGame.prodGame", "SafetyGame.maintains_prodGame", "SafetyGame.danger_le_of_rate", "SafetyGame.not_crossed_of_rate", "SafetyGame.consStream", "SafetyGame.playH", "SafetyGame.playH_positional", "SafetyGame.playH_shift", "SafetyGame.mem_recovery_of_recovers", "SafetyGame.mem_recovery_iff" ], "module": "AISafetyAtlas.Sovereignty.SafetyGame", "build_command": "lake build AISafetyAtlas.Sovereignty.SafetyGame", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Mixed for D1 -- the characterization is proved at arbitrary state, action and answer types where print assumes a finite game, and print's bound of |S| strict removal rounds on the downward iteration is NOT proved, so the object is obtained and the rate is not. Wider for D2: print states an equivalence and both directions are proved, with the converse quantified over history-dependent controllers rather than positional ones, so the recovery sets are where no controller of any kind succeeds from outside. Same for D3, D4 and D5. Wider for D4 in that the two hypotheses print states in prose -- independent action resources and factorized transitions -- are carried by the definition of the product game rather than assumed about a given one. Everything is positional and open-loop over an arbitrary adversary sequence; print's section 6.1 response order is the same. RELATED, not EXACT. Witnesses in AISafetyAtlas.Examples.Sovereignty.SafetyGame reach the root facade and are inside the axiom audit.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "ofTable", "keepsInside", "actionAt", "shrink", "shieldCandidate", "IsShield", "exists_action_of_isShield", "subset_safetyKernel_of_isShield", "exists_maintaining_of_isShield", "shieldAction" ], "module": "AISafetyAtlas.Sovereignty.ShieldCheck", "build_command": "lake build AISafetyAtlas.Sovereignty.ShieldCheck", "scope_delta": { "summary": "Runtime shield synthesis, and the one artifact in the checker family that returns something to install. The design point is that the fixpoint iteration is UNTRUSTED: it proposes a candidate envelope and subset_safetyKernel_of_isShield checks two decidable conditions on the result, which is sound whatever produced it, so no convergence proof is owed. exists_maintaining_of_isShield then turns membership into a positional controller holding for all time, through the kernel characterization. COMPLETENESS IS NOT CLAIMED -- a smaller envelope is a more conservative shield and not a wrong one, and an empty result is a limit of the search rather than a finding about the system. Backs atlas-check kind 'shield'. Witnessed in AISafetyAtlas.Examples.Sovereignty.Checkers at a certified envelope and at a system with none.", "evidence": "docs/provenance/practitioner-checkers.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.SafetyGame.mem_safetyKernel_iff_exists_maintaining", "type": "NEW_PROOF", "source_declarations": [ "SafetyGame", "SafetyGame.cpre", "SafetyGame.safetyKernel", "SafetyGame.subset_safetyKernel", "SafetyGame.recovery_reaches", "SafetyGame.maintains_prodGame", "SafetyGame.danger_le_of_rate" ], "application": "Runtime safety envelopes: whether a system can be kept inside its safe set forever is decided by the largest self-sustaining subset of it, and a certificate is that subset plus one permitted action at each of its states. Recovery is the same object read upward, and the budget is a rank the controller decreases. A safe set with an empty kernel means no positional controller exists, not that none has been found; history-dependent controllers are not covered by a theorem here." } ] }, "public": { "group": "Limits of control and regulation", "title": "The part of a safe region you can actually stay in", "summary": "Staying inside a safe set forever is possible exactly from the largest part of it that can sustain itself; the rest only looks safe, and getting back in has a budget that behaves like a countdown.", "use": "When a system has an operational envelope and a supervisor is meant to keep it inside, or to bring it back.", "attribution": "Atlas-side interpretation of an unpublished proposal; the fixed-point machinery is Mathlib's" } }, { "id": "LAND-SOV-BELIEF-001", "name": "Deciding on one's own beliefs, while another party writes them", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. C1 is a limitation ON a named published notion -- Rakow, A Doxastic Characterisation of Autonomous Decisive Systems, EPTCS 371 (2022) 103-119, arXiv:2209.14038, pinned privately, sha256 4acbf14ddc72ce2d7f0c8cd8895eb39d0ee1cb2571637a74881e8d45a622bfda and READ for this row, Definitions 3 to 5 and Table 1. Rakow's doxastic strategy s_d maps a history of beliefs to an action and the belief formation function maps histories of observations to beliefs; autonomous-decisive (Def. 5) requires s_d to follow a dominant, current-state decisive possible-worlds strategy at every observed history. WHAT IS MODELED. DoxasticAgent flattens that pair to one step: form : O -> Bel, dec : Bel -> Act reading nothing but the belief, and a belief-relative score with decisive as a FIELD, so the agent's consistency is discharged by construction and costs nothing -- which is print's proof, instantiate the two maps. channel is the two-party game form in which the message is the other party's move; the agent commits a move of the same type and it is read by nothing. CONTENT. sender_forces: the sender forces the action singleton at every message. mem_of_agent_forces: EVERY target the agent forces contains the whole range of dec composed with form, so the agent forces no proper part of it -- this is the general statement, not a countermodel. beliefDecisive_not_sovereign: at two messages with different actions the sender forces a singleton the agent does not. sender_forces_range reads the sender's control as exactly that range. WHAT THIS IS NOT. Not a defect in Rakow's problem, and the module says so: Rakow asks whether a system can act on the content of its beliefs at all and answers it relative to a FIXED belief formation, claiming nothing about whose belief formation it is. The scope difference runs both ways -- this module has no temporal logic, no goal list and no dominance ordering, so it does not state Rakow's notion; Rakow has no second party, so it does not state this. NOT VACUOUS. AISafetyAtlas.Examples.Sovereignty.Belief carries trusting, print's instantiation with both maps the identity: the sender installs either action and the agent guarantees neither, while being maximal at every belief it holds. B5 IN THE SAME MODULE, AND POINTING THE OTHER WAY. SignalExperiment is a finite prior with a signal law at each world, both probability vectors, which is the finite explicit model print states B5 in. sum_marginal_mul_posterior is print's identity: the posteriors weighted by the signal probabilities average back to the prior. Nothing needs the signal to have positive probability -- at a signal that cannot occur both the weight and the world's contribution are zero, so the ENNReal division convention is never relied on. eq_prior_of_constant_posterior is print's consequence: a belief held after EVERY signal that can occur is the prior, so an interested sender cannot install a different fixed posterior with probability one. C1 and B5 are not in tension and the module docstring says why: C1's agent takes its message AS its belief, B5's receiver knows the experiment generating the message and conditions on it. Print's own caveats -- unknown selection, deception about the experiment, misspecification, compromised update -- are different models and none is claimed here.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-AUDIT-001", "LAND-SOV-AUTH-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "DoxasticAgent", "DoxasticAgent.channel", "DoxasticAgent.sender_forces", "DoxasticAgent.mem_of_agent_forces", "DoxasticAgent.beliefDecisive_not_sovereign", "DoxasticAgent.sender_forces_range", "SignalExperiment", "SignalExperiment.marginal", "SignalExperiment.posterior", "SignalExperiment.sum_marginal", "SignalExperiment.sum_marginal_mul_posterior", "SignalExperiment.eq_prior_of_constant_posterior" ], "module": "AISafetyAtlas.Sovereignty.Belief", "build_command": "lake build AISafetyAtlas.Sovereignty.Belief", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Wider for C1 -- print exhibits one instantiation with both maps the identity, and mem_of_agent_forces is proved for EVERY doxastic agent, so the conclusion is not carried by the witness. Narrower against Rakow, deliberately and in the module docstring: no LTL goal list, no dominance ordering, no possible-worlds simulation, and belief formation is a function of the last message rather than of a history, so this does NOT state Rakow's Definition 5 and claims no coverage of it. The belief-relative score abstracts Rakow's simulation within the belief to its value, which is the minimum needed for decisive to mean anything and is less than Rakow's notion requires. For B5: Same as print, at print's finite explicit model, with the zero-probability signal handled rather than excluded.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.DoxasticAgent.mem_of_agent_forces", "type": "NEW_PROOF", "source_declarations": [ "DoxasticAgent", "DoxasticAgent.channel", "DoxasticAgent.sender_forces", "DoxasticAgent.beliefDecisive_not_sovereign" ], "application": "Assistants that act on a belief built from what they are told: an audit showing the system always takes the best action given its beliefs establishes nothing about who chooses the action, because every target such a system can guarantee already contains every action its message source can install. In the model the belief is a function of the message alone (the channel ignores the agent's move). Provenance of beliefs is a separate property from consistency with them, and only the first bears on control." } ] }, "public": { "group": "Limits of control and regulation", "title": "Believing what you are told is not deciding for yourself", "summary": "A system can always take the best action its beliefs allow and still control nothing, if somebody else writes the beliefs: everything such a system can guarantee already includes every action its message source can install.", "use": "When an assistant's autonomy is argued from its acting consistently on its own assessment of the situation.", "attribution": "Atlas-side interpretation of an unpublished proposal, which states this as a limitation on Rakow's doxastic notion of an autonomous-decisive system; no coverage of either is claimed" } }, { "id": "LAND-SOV-EVIDENCE-001", "name": "What a decision can achieve on the evidence it has", "tags": [ "decision-theory", "oversight" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. Built on the event gap of LAND-SOV-INFL-001. C7, THE COUNTING BOUND. actionLaw is the observation integrated out, Measure.bind, and actionLaw_congr is the only place the identical-observation hypothesis is used -- it IS print's step that the action distribution is the same in all contexts. exists_measure_le_inv_of_disjoint is then pure counting: k pairwise disjoint success sets under one probability law, so the smallest has measure at most 1/k. exists_success_le_inv puts the two together at print's own quantifiers -- k contexts, their observation laws, the controller, the success sets, all still binders -- so the scope grade is read off elaborated binders and not off a docstring. Randomization is not excluded and is not needed: the statement quantifies over an arbitrary law, and the deterministic case is the point mass. ATTAINED, NOT JUST BOUNDED. AISafetyAtlas.Examples.Sovereignty.Evidence carries print's symmetric construction -- uniform mass over k actions, one correct per context -- where every context succeeds with probability exactly 1/k, so the bound cannot be improved. C8, THE BINARY BOUND. success_add_le_one_add_eventGap is the undivided form: the chance of deciding 0 correctly plus the chance of deciding 1 correctly is at most one plus the gap. Print maximizes over the decision region by hand; eventGap ALREADY is that maximum, so the bound holds at every measurable region with no maximization step and the halving of print's statement is only the equal prior. eventGap_ge_of_success is print's contrapositive, written as twice the guaranteed success against one plus the gap so that no truncated subtraction appears. blind_success is the degenerate reading: identical laws give gap zero, and the two successes then sum to at most one. WHAT IS OUT OF SCOPE. Abstention. Print says a rule that may decline evades the bound only when the specification permits loss of service; a declining rule is not a decision region and is not modelled here.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-INFL-001", "LAND-SOV-AUDIT-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "actionLaw", "actionLaw_congr", "exists_success_le_inv", "exists_measure_le_inv_of_disjoint", "success_add_le_one_add_eventGap", "avgSuccess_le_one_add_eventGap", "eventGap_ge_of_success" ], "module": "AISafetyAtlas.Sovereignty.Evidence", "build_command": "lake build AISafetyAtlas.Sovereignty.Evidence", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Wider for C7 -- exists_success_le_inv carries print's binders (contexts, identical observation laws, controller, disjoint success sets) and the action space is an arbitrary measurable space with no finiteness anywhere, where print speaks of a finite action set. Wider for C8 -- the bound is proved at every measurable decision region rather than at the maximizing one, since the maximum is already the event gap. Narrower in one respect, stated in the module: abstention is outside both statements, as a declining rule is not a decision region; print names this caveat itself. The contrapositive is stated multiplicatively, 2c at most 1 plus the gap, rather than as print's TV at least 1 - 2 epsilon, so that ENNReal's truncated subtraction never appears.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.exists_success_le_inv", "type": "NEW_PROOF", "source_declarations": [ "actionLaw", "actionLaw_congr", "exists_measure_le_inv_of_disjoint", "success_add_le_one_add_eventGap", "eventGap_ge_of_success" ], "application": "Evaluating a monitor or a controller that cannot see which situation it is in: if several situations look the same and each demands a different response, the best achievable success is fixed by how many they are, and randomizing the response does not raise it. In the binary case the requirement is quantitative -- a target success level forces a minimum statistical distance between the situations' observable laws, which is a specification on the sensors, not on the decision rule." } ] }, "public": { "group": "Limits of control and regulation", "title": "A decision cannot beat the evidence it is given", "summary": "If several situations look identical and each needs a different response, success is capped by how many situations there are, and flipping a coin does not help; in the two-situation case the cap is set by how far apart the observations are statistically.", "use": "When setting a success target for a monitor or a controller that sees only part of the situation.", "attribution": "Atlas-side interpretation of an unpublished proposal; the total-variation reading is standard" } }, { "id": "LAND-SOV-COGSOV-001", "name": "The cognitive-sovereignty predicate, and what belief change does not prove", "tags": [ "multi-agent", "verification" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. CogSov TRANSCRIBES section 7.4's boxed predicate: one defender D in the admissible class whose composed runs lie in the relational specification AND whose every observation of every authorized request in every environment lies inside the authorized trace language intersected with the required service. EVERY COMPONENT IS A PARAMETER and the module says so: the definition carries no content by itself. What it fixes is the SHAPE -- the SAME D must provide the service and satisfy the authorship condition, and the authorship condition is a property of a SET of runs, which is why H has type Set (Set Run) and not Set Run. Print's own caveat that a constitution's legitimacy is a separate assumption is quoted and nothing here claims otherwise. TWO CHECKS A DEFINITION WITH THIS MANY PARAMETERS OWES. cogSov_holds exhibits a defender meeting both clauses, so the predicate is satisfiable; not_cogSov_of_service_failure exhibits a system whose only admissible defender emits an observation outside the required service, so the predicate is not satisfied by everything. Without both, an abstract definition is not worth stating. C4 IS THE SEPARATION, NOT THE DEFINITION. magnitude_does_not_decide_authorship gives two protected transitions with the SAME belief at every endorsed evidence and every request, one residual-free and one not. The endorsed influence is therefore identical and can be maximal, while only the first satisfies C3's endorsement condition of LAND-SOV-AUDIT-001. So no function of the magnitude of belief change separates the authorized case from the unauthorized one, which is print's conclusion that magnitude is not an authorship test. Print's other two hypotheses, authenticated requests and intact alternatives, are the two clauses of CogSov and are NOT what the separation turns on -- that is a modelling claim and it is recorded here rather than hidden. TRANSPORT, 2026-09-13. H's type was Set (Set Run) and is now AISafetyAtlas.Compositional.Hyperproperties.Hyperproperty at Run. That abbreviation is reducible, so no statement changed and no proof changed; what changed is that the Sovereignty and Compositional clusters now name one object, which the module docstring had recorded as NOT done and as a signature change. WHAT THE TRANSPORT BUYS: cogSov_mono_of_subsetClosed. If the relational specification is subset-closed - Clarkson and Schneider's page-1167 class, in the tree as Compositional.Hyperproperties.SubsetClosed since 2026-09-13, with their Theorem 1 as subsetClosed_of_isHyperSafetyOp - then DELETING RUNS CANNOT DESTROY COGNITIVE SOVEREIGNTY, with the same defender. AND THE HYPOTHESIS IS LOAD-BEARING: Examples.Sovereignty.CogSov.topOnly is the specification that accepts the total run set and nothing less, topOnly_not_subsetClosed shows it is outside the class, and shrinking_can_destroy_cogSov flips the predicate by shrinking the run set alone. STILL NOT DECIDED, and the docstring still says so: whether H OUGHT to be required to be subset-closed. The transport makes the question askable inside the tree and does not answer it.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-AUDIT-001", "LAND-SOV-BELIEF-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "CogSov", "cogSov_of_witness", "not_cogSov", "cogSov_mono_of_subsetClosed", "magnitude_does_not_decide_authorship" ], "module": "AISafetyAtlas.Sovereignty.CogSov", "build_command": "lake build AISafetyAtlas.Sovereignty.CogSov", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for the section 7.4 definition, at arbitrary types for implementations, defenders, requests, environments, runs and observations, with the probability-threshold variant print mentions NOT carried. Wider for C4 -- print asserts compatibility at one model, and magnitude_does_not_decide_authorship is proved for an arbitrary endorsed map and an arbitrary pair of unendorsed inputs, so the separation is not carried by the witness. Narrower in one respect stated in the module: print derives C4 from authenticated requests, intact alternatives and no residual unendorsed dependence together, and only the third is what the separation turns on here; the first two are the clauses of CogSov and are not consumed.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.magnitude_does_not_decide_authorship", "type": "NEW_PROOF", "source_declarations": [ "CogSov", "cogSov_of_witness", "not_cogSov" ], "application": "Reviewing whether an assistant's influence on a person's beliefs was legitimate: how much the belief moved is not evidence either way, because a transition that depends only on endorsed evidence and one that also depends on an unendorsed channel can produce exactly the same change. What distinguishes them is the dependence structure, which is what an endorsement interface is for." } ] }, "public": { "group": "Limits of control and regulation", "title": "How much a belief moved says nothing about who moved it", "summary": "Two systems can change someone's beliefs by exactly the same amount, one using only evidence the person endorsed and one also using a channel they did not; so the size of the change cannot tell them apart.", "use": "When judging whether an assistant's influence on a user's beliefs was legitimate.", "attribution": "Atlas-side interpretation of an unpublished proposal, whose section 7.4 definition is transcribed and whose compatibility claim is strengthened to a separation" } }, { "id": "LAND-SOV-EMPOWER-001", "name": "Empowerment of a deterministic channel is the capacity of its range", "tags": [ "information-theory", "control-theory" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. B2 IS THE ONE RESULT THAT JOINS THIS CLUSTER TO THE ATLAS'S INFORMATION THEORY. empowerment is the supremum of I(input ; output) over ALL input laws, written as a sSup of a set of reals rather than an iSup over a type of measures so that no typeclass is carried into the definition. CONTENT. mutualInfo_id_eq_entropy is print's first step: a deterministic output has no conditional entropy given the input, so the mutual information IS the output entropy -- proved through PFR's chain_rule'' and entropy_prod_comp, not assumed. mutualInfo_le_log_card_range is the bound, PFR's entropy_le_log_card_of_mem_finite at the range. ATTAINMENT IS PRINT'S OWN CONSTRUCTION and is carried out, not asserted: section' is one input per reachable output via Set.rangeSplitting, isUniform_section shows the uniform law on it makes the OUTPUT uniform on the range (the computation is that the section meets each fibre exactly once), and IsUniform.entropy_eq' closes it. empowerment_eq_log_card_range is B2. THE BRIDGE. empowerment_eq_channelCapacity_range identifies the answer with AISafetyAtlas.InformationTheory.channelCapacity of the reachable outputs. The two definitions were written for different purposes -- one counts signals a noiseless channel carries, the other maximizes an information over input laws -- and they agree here. PRINT'S CAVEAT IS CARRIED AS A CAVEAT. With a stochastic channel, an adversary, or constraints on the allowable input distributions, the reachable-output count alone is insufficient; the supremum here is over all input laws, so the constrained case is outside it, and mutualInfo_le_log_card_range is the part that survives a restriction. NOT VACUOUS. AISafetyAtlas.Examples.Sovereignty.Empowerment: an actuator with four settings that only flips one bit has exactly one bit of empowerment, and a constant actuator has none whatever its input alphabet.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-COMM-001", "LAND-SOV-INFL-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; PFR as pinned; atlas IN_TREE", "declarations": [ "section'", "section'_inter_preimage", "isUniform_section", "condEntropy_comp_id_eq_zero", "mutualInfo_id_eq_entropy", "mutualInfo_le_log_card_range", "mutualInfo_section_eq", "empowerment", "empowerment_eq_log_card_range", "empowerment_eq_channelCapacity_range" ], "module": "AISafetyAtlas.Sovereignty.Empowerment", "build_command": "lake build AISafetyAtlas.Sovereignty.Empowerment", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for B2, with the attainment carried out rather than asserted -- print says choose one input per reachable output and make those outputs equiprobable, and section' with isUniform_section is that construction. Wider in that the output type needs only to be finite and discretely measurable and the supremum is over arbitrary probability measures on the input, where print speaks of a finite channel and input distributions. Narrower exactly where print says it should be: the stochastic channel, the adversarial case and the constrained-input case are not stated, and the module records that mutualInfo_le_log_card_range is the half that survives a restriction on allowable inputs.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.empowerment_eq_log_card_range", "type": "NEW_PROOF", "source_declarations": [ "empowerment", "mutualInfo_id_eq_entropy", "mutualInfo_le_log_card_range", "isUniform_section", "empowerment_eq_channelCapacity_range" ], "application": "Sizing an actuator or an interface by what it can actually reach: the information a deterministic effector can carry about its own inputs is fixed by the number of distinct outcomes it can produce, so adding settings that collapse to the same outcome adds nothing. The bound is also the right one to quote when the question is how much of its own state an agent can signal through a fixed deterministic mechanism." } ] }, "public": { "group": "Limits of control and regulation", "title": "An actuator is worth what it can reach", "summary": "How much a system can influence through a deterministic mechanism depends only on how many distinct outcomes that mechanism produces, not on how many settings it has.", "use": "When sizing an effector, an interface, or any deterministic channel by its actual influence.", "attribution": "Atlas-side interpretation of an unpublished proposal; the entropy machinery is PFR's" } }, { "id": "LAND-SOV-MINIMAX-001", "name": "Mixed strategies close the gap a pure commitment leaves", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. Q3 WAS RECORDED AS BLOCKED AND WAS NOT. The earlier search asked whether the pinned Mathlib has a LINEAR PROGRAM -- it does not, the convex-cone files carry a Farkas-shaped dual and nothing named a linear program -- and concluded Q3 needed an external development. It asked the wrong question: Mathlib.Topology.Sion carries Sion's form of the von Neumann minimax theorem at the pinned revision, and Sion.exists_isSaddlePointOn is stated for real-valued functions. Print reaches Q3 by LP strong duality; this reaches the same STATEMENT by convexity. A different route to the same statement is not a scope difference. CONTENT. mixed k is the standard simplex, so nonempty (given a pure strategy), convex and compact come from Mathlib. payoff is the bilinear expectation; payoffLeft and payoffRight package it as linear maps so that convexity and concavity -- hence quasiconvexity and quasiconcavity -- are LinearMap.convexOn and LinearMap.concaveOn rather than hand computations, and continuity is fun_prop. exists_mixed_saddlePoint is Sion instantiated. exists_mixed_value is the form Q3 is used in: a value v, a row mixture capping the payoff at v against every reply, and a column mixture holding it at v or above. That IS the coincidence of the two security levels and it is stated WITHOUT a supremum, so no conditional-completeness bookkeeping enters and no boundedness hypothesis is needed. NOT VACUOUS, AND THE GAP IS EXHIBITED. AISafetyAtlas.Examples.Sovereignty.Minimax carries print's simultaneous-bit game as the identity payoff matrix: pure_security_zero shows every pure commitment has a reply scoring zero, half_value shows the even mixture scores exactly one half against every reply from either side. Pure 0, mixed 1/2, which is print's own numbers. UPSTREAM NOTE. Mathlib's own Sion.lean lists spelling out the von Neumann case as a TODO, so the finite bilinear instance is not there. ORIENTATION IS NOT A CONVENTION. Mathlib's IsSaddlePointOn X Y f a b reads for all x in X, for all y in Y, f a y at most f x b, so its X is the MINIMIZING side. payoff is the ROW player's, and print has the row player maximizing, so the column player is X in the instantiation and the row player is Y. The first version of this module had them the other way, which is still provable and is NOT Q3: it certifies as the value the number the row player is held to rather than the number it can guarantee. A symmetric payoff matrix cannot tell the two apart, which is why the examples now carry an asymmetric one -- skewed = [1, 2] with one row -- where skewed_value gives 1 from both sides and skewed_not_two refutes the number the reversed reading would have certified.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-VALUE-001", "LAND-SOV-INFL-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "mixed", "payoff", "payoffLeft", "payoffRight", "mixed_nonempty", "mixed_convex", "mixed_isCompact", "continuous_payoffLeft", "continuous_payoffRight", "exists_mixed_saddlePoint", "exists_mixed_value" ], "module": "AISafetyAtlas.Sovereignty.Minimax", "build_command": "lake build AISafetyAtlas.Sovereignty.Minimax", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for Q3's statement, reached by a different route -- Sion's theorem rather than linear-programming strong duality, which the pinned Mathlib cannot supply. Wider in that no bound on the payoff matrix is assumed anywhere: print states boundedness, and finiteness of the pure strategy sets already gives it. Narrower in one respect and deliberately: the two programs print writes out, the row player's and its dual, are NOT formalized, so the LP certificate view of the value is absent and only the saddle point and the two security levels are proved. The value is stated as a coincidence of security levels rather than as an equality of a supremum and an infimum, which avoids conditional completeness in the reals and is the form the result is consumed in.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.exists_mixed_value", "type": "NEW_PROOF", "source_declarations": [ "mixed", "payoff", "exists_mixed_saddlePoint" ], "application": "Adversarial evaluation with a finite action set on both sides: a defender that commits to a single deterministic response can be answered, and the guarantee it can actually hold is the mixed one. The theorem says a randomized commitment achieving the value always exists, so a security level computed over deterministic policies understates what is achievable, and the simultaneous-bit example is the smallest case where it understates it by the whole gap." } ] }, "public": { "group": "Limits of control and regulation", "title": "Randomizing is worth something exact", "summary": "In a finite two-party contest, committing to one fixed action can be answered and guarantees little, while randomizing guarantees a definite amount that neither side can improve on.", "use": "When a defender's guarantee is computed over deterministic policies only.", "attribution": "Atlas-side interpretation of an unpublished proposal; the minimax theorem is Mathlib's formalization of Sion after Komiya" } }, { "id": "LAND-SOV-OUTNUMBERED-001", "name": "Being outnumbered is not being outmatched", "tags": [ "multi-agent", "oversight", "control-theory" ], "notes": "GROUNDING. This row is NOT from the unpublished proposal and claims no source: it is this repository's own reading of a question raised on 2026-09-12 -- why one guard holds a prison full of prisoners who could overwhelm him, and what that means for many parties who nominally hold override authority over one system. The question is recorded as asked and the answer is recorded as interpretation. THREE MECHANISMS, TWO OF THEM HERE. First, it is NOT in the game form. Forces quantifies over a JOINT commitment, so coordination is free and the many simply win; forces_superadditive says merging never costs. That is not a defect -- at a genuinely simultaneous move the outnumbered guard does lose, and the model says so. What a game form can say is that the win needs everyone: not_forces_of_veto_outside shows a coalition disjoint from a vetoing party forces nothing, so one absentee kills unanimity however large the coalition. Second, it IS in the schedule. AdversaryLe says every move of one adversary is matched by a move of another; cpre_subset_of_adversaryLe and safetyKernel_subset_of_adversaryLe say a stronger adversary keeps less. The guard's position is that this inclusion is STRICT between the adversary who may change one coordinate per round and the adversary who may change all at once -- same members, same reach, different schedule. Strictness is a property of a system and not a theorem, so it is a witness: AISafetyAtlas.Examples.Sovereignty.Outnumbered carries revolt, where the state counts those in revolt, the guard puts down one per round, and the adversary adds up to r. At r = 1 the guard holds the threshold forever from a quiet start with no horizon bound; at r = 2 it loses in ONE round from the same start, and no action helps. kernel_ssubset is the strict inclusion. THIRD MECHANISM IS ABSENT AND SAID TO BE ABSENT. The profile that overthrows the guard is not an equilibrium -- whoever moves first is punished, so nobody moves, so the coalition's effectivity is never exercised. That needs preferences over outcomes and a deviation-stability notion, and this repository has neither. The module docstring says so rather than leaving a reader to assume the result covers it. MODELLING CHOICE. The unanimity game form has outcome type Prop rather than Bool, so no decidability instance is needed and the known non-reduction of an exposed definition at instance synthesis cannot bite.", "original_source_refs": [], "related_result_ids": [ "LAND-SOV-SAFETYGAME-001", "LAND-SOV-AUTH-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "not_forces_of_veto_outside", "AdversaryLe", "cpre_subset_of_adversaryLe", "safetyKernel_subset_of_adversaryLe" ], "module": "AISafetyAtlas.Sovereignty.Outnumbered", "build_command": "lake build AISafetyAtlas.Sovereignty.Outnumbered" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.safetyKernel_subset_of_adversaryLe", "type": "NEW_PROOF", "source_declarations": [ "AdversaryLe", "cpre_subset_of_adversaryLe", "not_forces_of_veto_outside" ], "application": "Human oversight of a deployed system, where many parties each hold a nominal override and the system is one. If the overrides can only be exercised one at a time, the set of states from which the system can be kept inside its envelope contains, and can be strictly larger than (kernel_ssubset in the example), the set when they could be exercised together -- and the difference is the whole of the system's practical autonomy, with no change in anyone's formal authority. The same reading applies to a vendor and the institutions depending on it." } ] }, "public": { "group": "Limits of control and regulation", "title": "Why the outnumbered side wins", "summary": "When many parties could together override one system but can only act one at a time, the system keeps control of situations it would lose if they acted together; nobody's authority changed, only the timing.", "use": "When an oversight arrangement gives many people a veto that they would have to exercise together.", "attribution": "Atlas-original reading of a question about one-to-many power; not from any source, and the incentive side of the question is explicitly not addressed" } }, { "id": "LAND-VERIF-AGENTBEHAVIOR-001", "name": "No total verifier for a nontrivial behavioural safety specification", "tags": [ "oversight", "verification" ], "notes": "GROUNDING. Melo, Maximo, Soma and Castro, On the Undecidability of Artificial Intelligence Alignment: Machines that Halt, arXiv:2408.08995v1, read from rendered pages 1, 3 and 4. Graded in section 24 of docs/provenance/source-coverage-audit.md, the first section that source has had and the first registry row this module has had. WHY THIS ROW EXISTS SEPARATELY FROM BY-012. BY-012 is the upstream-reuse row for Rice's theorem: its formalizations are Mathlib's ComputablePred.rice and Isabelle's Rice_2, and it names this module's theorem only in its lean_artifact declaration list, as a consumer. That is the PROOF ROUTE. This row is the STATEMENT source. The module's header previously had the two reversed, naming Rice as source and Melo as related literature; it is corrected. CONTENT. no_behavioral_safety_verifier is print's claim at print's quantifiers. Agent is a program code with partial-recursive I/O semantics, which is print's own partial-function reading. SafetySpec and Satisfies make print's extensionality condition structural rather than argued. SpecNontrivial is print's non-triviality, and print supplies both witnesses rather than only the adjective. BehavioralSafetyVerifier unpacks print's word 'decides' into total, computable, and correct in both directions - more explicit than print and claiming nothing more. PROOF. Reduces through the atlas's rice packaging of Mathlib's Rice. Print offers that route first and an explicit Halting-Problem reduction second; the atlas does not take the second and builds no halting bridge. rice_code_iff is the semantic-to-code crossing that print's informality steps over, and is graded Beyond. NOT HERE: print's enumerable set of architecturally aligned models (an architecture argument with no enumeration exhibited and no closure proved), its special decidable case for finite-input networks, its masking architecture, and its proposed halting constraint on the judge. RICE IS CITED AND NOT PINNED: ams.org returns HTTP 403 to automated requests and no workaround was attempted. No grade depends on it, since the atlas formalizes none of Rice's statements and consumes Mathlib's instead. Recorded in the private manifest of 2026-09-11 under 'Named, not pinned'.", "original_source_refs": [ "atlas-ref-melo-2024", "survey-ref-037" ], "related_result_ids": [ "BY-012" ], "relations": [ { "target": "BY-012", "kind": "BUILDS_ON", "note": "no_behavioral_safety_verifier reduces to the atlas's rice packaging of Mathlib's ComputablePred.rice and reproves nothing. BY-012 is the row for that route; this row is the row for the statement Melo et al. make." } ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Agent", "SafetySpec", "Satisfies", "SpecNontrivial", "BehavioralSafetyVerifier", "no_behavioral_safety_verifier" ], "module": "AISafetyAtlas.Verification.AgentBehavior", "build_command": "lake build AISafetyAtlas.Verification.AgentBehavior", "relationship": "RELATED", "scope_delta": { "summary": "Print's undecidability claim at print's own quantifiers, with print's own partial-function objects, print's extensionality made structural, and print's word 'decides' unpacked into total, computable and correct both ways. No widening: the extra precision unpacks print's words rather than weakening its hypotheses. RELATED rather than EXACT because print numbers nothing, because the atlas takes print's Rice route and not its Halting-Problem reduction, and because print's second half - the enumerable set of architecturally aligned models, the finite-input decidable case, and the halting constraint on the judge - is absent.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Verification.AgentBehavior.no_behavioral_safety_verifier", "type": "NEW_PROOF", "source_declarations": [ "no_behavioral_safety_verifier" ], "application": "AI oversight: no total, computable, sound-and-complete procedure classifies every encoded agent against a nontrivial extensional input/output safety specification. Melo, Maximo, Soma and Castro (arXiv:2408.08995v1), abstract and Formal Proof section; proved by reduction to Rice via Mathlib." } ] }, "public": { "group": "What an observer can recover", "title": "No general behavioural safety checker", "summary": "No program can read another program and always say correctly whether its input/output behaviour is acceptable.", "use": "When a proposal assumes an automatic checker for arbitrary models.", "attribution": "Melo, Maximo, Soma & Castro; proved via Rice's theorem" } }, { "id": "LAND-LINSYS-001", "name": "Kalman and Hautus: when a linear system's state is determined, and when it can be driven", "tags": [ "control-theory" ], "notes": "GROUNDING. The classical algebraic criteria for a finite-dimensional linear time-invariant system: the Kalman rank conditions and the Hautus eigenvalue tests, with the duality between the observability and controllability sides. The Lean is adapted from AnandGokhale/LeanForControl at commit c5cedca (Apache-2.0, gokhale-leanforcontrol-2026); the printed sources are named in the row's source refs and are NOT PINNED - see the private manifest of 2026-09-11 in the literature directory. NOTHING HERE IS GRADED against a printed statement and no coverage-audit section claims one. WHAT IS PROVED. IsObservable is the condition that no nonzero state is annihilated by every C * A^k for k < n, and IsControllable that every state is a combination of the columns of A^k * B. Each is equivalent to its Kalman rank condition, observability is equivalent to triviality of the unobservable subspace, and over the complex numbers each is equivalent to its Hautus pencil condition at every eigenvalue candidate. isControllable_iff_isObservable_transpose is the duality, and the controllability Hautus test is derived through it. THE LIMIT THAT MATTERED, AND NO LONGER DOES (falsified 2026-09-20, retracted here 2026-09-21). This field used to read: 'This layer is algebraic and nothing else. No declaration here defines a trajectory, a solution of the differential equation, or an output signal, so the reading the state cannot be reconstructed from the outputs is NOT proved: the bridge from output trajectories to the Kalman condition is not in the tree.' AISafetyAtlas.LinearSystems.Dynamics defines IsTrajectoryOn and outputSignal and AISafetyAtlas.LinearSystems.Flow builds the matrix exponential, variation of constants and the adjoint argument. DeterminesStateOn and IsCompletelyReachable are the two properties themselves, and determinesStateOn_iff_isObservable and isCompletelyReachable_iff_isControllable make each criterion EQUIVALENT to the property named for it - the observability one on any non-empty open window. The survey rows BY-001 and BY-002 were promoted to covered on 2026-09-21 on the strength of it, and they own the two equivalences; this row keeps the algebraic criteria and the modules. NOT KLAMKA. survey-ref-021, the source those two rows cite, is described in catalogue as giving a MINIMAL-POLYNOMIAL SUFFICIENT condition for uncontrollability and unobservability. The criteria here are CHARACTERIZATIONS through the Kalman matrices and the Hautus pencils, so they are a neighbouring development rather than that statement - CONFIRMED 2026-09-11 against the two printed pages, now pinned: print's Theorem 1, Theorem 2 and four corollaries are one-directional Jordan-block counts against the rank of B or C, and the note states neither criterion carried here. THE SAME DAY the Hautus half was proved: all six conclusions now hold in the tree under a hypothesis on the eigenspace dimension, which is strictly weaker than print's. They are graded Partial and not Yes because the bridge from print's antecedent to that one, print's eq (5), is not proved here. Graded in section 25 of docs/provenance/source-coverage-audit.md. Full triage: docs/provenance/by001-by002-linear-systems-triage.md. NOT HERE. Time-varying systems, discrete-time reachability over a ring, stabilizability and detectability, the Kalman decomposition, minimal realizations, the Jordan canonical form, and any statement about a nonlinear system. 'And every dynamical statement' stood in this list until 2026-09-21 and is struck: the two properties print reasons about are now here. OWNERSHIP MOVED 2026-09-21. The two equivalences this row briefly listed - determinesStateOn_iff_isObservable and isCompletelyReachable_iff_isControllable - are now owned by the survey claim rows BY-001 and BY-002, which they cover, and a declaration is owned by one row. This row keeps the algebraic criteria and the modules; the two claim rows keep the bridge to the dynamical properties.", "original_source_refs": [ "hautus-1969-controllability-observability", "kalman-1963-linear-dynamical-systems", "gokhale-leanforcontrol-2026" ], "related_result_ids": [], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "module": "AISafetyAtlas.LinearSystems", "build_command": "lake build AISafetyAtlas.LinearSystems", "relationship": "RELATED", "scope_delta": { "summary": "The aggregating facade. It declares nothing of its own and imports the seven modules below; its docstring carries the map and the provenance. ITS CENTRAL CLAIM WAS RETRACTED 2026-09-21, a day after the work that falsified it: it used to state that this layer is algebraic and proves no dynamical claim, and Dynamics and Flow now define the differential equation, its solutions and the output signal and prove each criterion equivalent to the property named for it.", "evidence": "docs/provenance/by001-by002-linear-systems-triage.md" }, "declaration": "AISafetyAtlas.LinearSystems.isObservable_iff_hautus" }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "mulVec_kernel_trivial_iff_rank_eq_card_cols", "mulVec_range_top_iff_rank_eq_card_rows", "rank_fromCols_le", "rank_fromRows_le" ], "module": "AISafetyAtlas.LinearSystems.MatrixLemmas", "build_command": "lake build AISafetyAtlas.LinearSystems.MatrixLemmas", "relationship": "RELATED", "scope_delta": { "summary": "Two rank-nullity bridges between the kernel and rank forms for a matrix acting by mulVec. Support lemmas for the criteria, not statements of any source. ADDED 2026-09-11, atlas-original and not from the upstream development: rank_fromCols_le and rank_fromRows_le, that a block matrix has no more rank than its blocks apart. Domain-neutral, over an arbitrary field, written for the Klamka bound in the Hautus module.", "evidence": "docs/provenance/by001-by002-linear-systems-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "controllabilityMatrix", "controllabilityMatrix_apply", "IsControllable", "isControllable_iff_controllabilityMatrix_rank_eq" ], "module": "AISafetyAtlas.LinearSystems.Controllability", "build_command": "lake build AISafetyAtlas.LinearSystems.Controllability", "relationship": "RELATED", "scope_delta": { "summary": "Standard criteria at textbook generality: the definitions over a semiring, the rank equivalences over a field, and the Hautus tests over the complex numbers, where the eigenvalue argument needs algebraic closure. RELATED rather than EXACT because the printed sources are named and not read: no statement in this row has been compared with a page. The atlas adds no mathematics to the upstream development; the changes are the namespace, the module marker and public visibility, removal of the Architect dependency and its blueprint attributes, and two repairs forced by the Mathlib bump - a haveI the style linter rejects and one simpa inside exists_eigenvector_of_unobservableSubspace_neBot, both marked at the site.", "evidence": "docs/provenance/by001-by002-linear-systems-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "observabilityMatrix", "IsObservable", "observabilityMatrix_apply", "observabilityMatrix_mulVec_apply", "isObservable_iff_observabilityMatrix_ker_trivial", "isObservable_iff_observabilityMatrix_rank_eq" ], "module": "AISafetyAtlas.LinearSystems.Observability", "build_command": "lake build AISafetyAtlas.LinearSystems.Observability", "relationship": "RELATED", "scope_delta": { "summary": "Standard criteria at textbook generality: the definitions over a semiring, the rank equivalences over a field, and the Hautus tests over the complex numbers, where the eigenvalue argument needs algebraic closure. RELATED rather than EXACT because the printed sources are named and not read: no statement in this row has been compared with a page. The atlas adds no mathematics to the upstream development; the changes are the namespace, the module marker and public visibility, removal of the Architect dependency and its blueprint attributes, and two repairs forced by the Mathlib bump - a haveI the style linter rejects and one simpa inside exists_eigenvector_of_unobservableSubspace_neBot, both marked at the site.", "evidence": "docs/provenance/by001-by002-linear-systems-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "unobservableSubspace", "mem_unobservableSubspace_iff", "unobservableSubspace_eq_bot_iff_isObservable", "A_mulVec_mem_unobservableSubspace_of_mem", "hautusObservabilityMatrix", "hautusObservabilityMatrix_mulVec_eq_zero_iff", "not_isObservable_implies_hautus_failure", "hautus_failure_implies_not_isObservable", "isObservable_iff_hautus", "controllabilityMatrix_transpose", "isControllable_iff_isObservable_transpose", "hautusControllabilityMatrix", "hautusControllabilityMatrix_transpose", "isControllable_iff_hautus", "not_isControllable_of_rank_add_lt", "finrank_ker_add_rank_eq", "not_isControllable_of_finrank_ker_gt_rank", "not_isControllable_of_finrank_ker_gt_width", "finrank_ker_transpose_eq", "not_isObservable_of_finrank_ker_gt_rank", "not_isObservable_of_finrank_ker_gt_height", "not_isControllable_and_not_isObservable_of_finrank_ker_gt_ranks", "not_isControllable_and_not_isObservable_of_finrank_ker_gt_dims" ], "module": "AISafetyAtlas.LinearSystems.Hautus", "build_command": "lake build AISafetyAtlas.LinearSystems.Hautus", "relationship": "RELATED", "scope_delta": { "summary": "Standard criteria at textbook generality: the definitions over a semiring, the rank equivalences over a field, and the Hautus tests over the complex numbers, where the eigenvalue argument needs algebraic closure. RELATED rather than EXACT because the printed sources are named and not read: no statement in this row has been compared with a page. The atlas adds no mathematics to the upstream development; the changes are the namespace, the module marker and public visibility, removal of the Architect dependency and its blueprint attributes, and two repairs forced by the Mathlib bump - a haveI the style linter rejects and one simpa inside exists_eigenvector_of_unobservableSubspace_neBot, both marked at the site. KLAMKA SECTION, ADDED 2026-09-11 and atlas-original: the nine declarations from not_isControllable_of_rank_add_lt onwards carry the conclusions of Klamka's Theorem 1, Theorem 2 and Corollaries 1 to 4 under a hypothesis on the eigenspace dimension rather than print's minimal-polynomial quantity. Graded Partial and Wider in section 25 of the coverage audit: the atlas hypothesis is strictly weaker, and the inequality that would let print's antecedent reach it - print's eq (5) - is not in the tree.", "evidence": "docs/provenance/by001-by002-linear-systems-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.LinearSystems.finrank_ker_comp_le", "AISafetyAtlas.LinearSystems.finrank_ker_pow_le", "AISafetyAtlas.LinearSystems.finrank_maxGenEigenspace_le_mul_of_stabilizes", "AISafetyAtlas.LinearSystems.finrank_maxGenEigenspace_le_index_mul", "AISafetyAtlas.LinearSystems.rootMultiplicity_le_mul_finrank_eigenspace", "AISafetyAtlas.LinearSystems.ceil_div_le_finrank_eigenspace", "AISafetyAtlas.LinearSystems.eigenspace_mulVecLin_eq_ker", "AISafetyAtlas.LinearSystems.charpoly_mulVecLin", "AISafetyAtlas.LinearSystems.not_isControllable_of_ceil_div_gt_rank", "AISafetyAtlas.LinearSystems.not_isControllable_of_ceil_div_gt_width", "AISafetyAtlas.LinearSystems.not_isObservable_of_ceil_div_gt_rank", "AISafetyAtlas.LinearSystems.not_isObservable_of_ceil_div_gt_height" ], "module": "AISafetyAtlas.LinearSystems.BlockBound", "atlas_module": "AISafetyAtlas.LinearSystems.BlockBound", "build_command": "lake build AISafetyAtlas.LinearSystems.BlockBound", "relationship": "RELATED", "scope_delta": { "summary": "ATLAS-ORIGINAL, and the module that closes the gap section 25 of the coverage audit named. Klamka's Theorem 1, Theorem 2 and Corollaries 1 and 2 are now stated at PRINT'S OWN ANTECEDENT - the multiplicity of the eigenvalue in the characteristic polynomial divided by an exponent at which the generalized eigenspace chain has stabilized, rounded up - rather than only at the eigenspace dimension. The counting step print derives from the Jordan form is derived here without one: finrank_ker_comp_le says the kernel of a composite is no bigger than the two kernels together, and finrank_ker_pow_le is its induction, that the kernel of a k-th power is at most k times the kernel. NEITHER IS IN MATHLIB at the pinned revision, and neither mentions an eigenvalue; a search for finrank of a kernel of a power and for a kernel bound on a composite returned nothing. WIDER THAN PRINT ON ONE AXIS: print states the test at the index, the LEAST stabilizing exponent, and these are stated at an ARBITRARY stabilizing exponent, which is a weaker hypothesis and therefore a stronger theorem; the index form is the immediate corollary through Mathlib's maxGenEigenspace_eq. THE AXIS THAT WAS NARROWER CLOSED 2026-09-20: print's index is the multiplicity of the eigenvalue in the MINIMAL polynomial and the exponent here was characterized by stabilization of the generalized eigenspace chain, and the identification print cites Zadeh and Desoer for is now proved - maxGenEigenspaceIndex_eq_rootMultiplicity_minpoly, in the two halves print's wording asks for: maxGenEigenspace_eq_genEigenspace_rootMultiplicity_minpoly says the chain has stopped at that multiplicity and rootMultiplicity_minpoly_le_of_stabilizes says no smaller exponent stops it. NOT IN MATHLIB at the pinned revision, measured before building: Mathlib relates a ROOT of the minimal polynomial to an eigenvalue and nothing relates the multiplicity to the index. NO JORDAN FORM, NO ALGEBRAIC CLOSURE, NO SPLIT MINIMAL POLYNOMIAL, and over an ARBITRARY FIELD where print is over the complex numbers: factor psi = (X - lambda)^nu * q and use Bezout for one half and minpoly.dvd for the other. ALL SIX NUMBERED RESULTS NOW ALSO EXIST WITH PRINT'S OWN nu: not_isControllable_of_ceil_div_minpoly_gt_rank and _gt_width, not_isObservable_of_ceil_div_minpoly_gt_rank and _gt_height, and the two conjunctions not_isControllable_and_not_isObservable_of_ceil_div_minpoly_gt_ranks and _gt_dims, where the exponent is computed from A through minpoly_mulVecLin rather than supplied by the caller. THE CONJUNCTIONS WERE HELD BACK BY A COST THAT DID NOT EXIST: the audit said they need a stabilizing exponent for A and for its transpose at once, and they do not - finrank_ker_transpose_eq collapses the two eigenspace dimensions into one, so print's antecedent on A alone reaches both halves, in four lines. NO JORDAN FORM anywhere, and Mathlib carries none at this revision. NOT VACUOUS: Examples.LinearSystems.Criteria.frozenC_stabilizes and frozenC_rootMultiplicity inhabit the hypothesis at the identity on two states, and frozenC_not_isControllable_klamka fires the test at print's own quantities. THE INDEX IS WITNESSED TOO: frozenC_minpoly_rootMultiplicity gives print's nu = 1 and frozenC_maxGenEigenspaceIndex the stabilization index at the same number, and frozenC_not_isControllable_klamka_minpoly and frozenC_not_isObservable_klamka_minpoly fire the test with both multiplicities read off A.", "evidence": "docs/provenance/by001-by002-linear-systems-triage.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "AISafetyAtlas.LinearSystems.IsTrajectoryOn", "AISafetyAtlas.LinearSystems.IsTrajectory", "AISafetyAtlas.LinearSystems.isTrajectoryOn_univ_iff", "AISafetyAtlas.LinearSystems.outputSignal", "AISafetyAtlas.LinearSystems.DeterminesStateOn", "AISafetyAtlas.LinearSystems.DeterminesInitialState", "AISafetyAtlas.LinearSystems.hasDerivAt_mulVec", "AISafetyAtlas.LinearSystems.IsTrajectoryOn.hasDerivAt_sub", "AISafetyAtlas.LinearSystems.mulVec_pow_eq_zero_of_outputSignal_eq_zero", "AISafetyAtlas.LinearSystems.isObservable_imp_determinesStateOn", "AISafetyAtlas.LinearSystems.isObservable_imp_determinesInitialState", "AISafetyAtlas.LinearSystems.eigenTrajectory", "AISafetyAtlas.LinearSystems.eigenTrajectory_isTrajectoryOn", "AISafetyAtlas.LinearSystems.not_determinesStateOn_of_not_isObservable", "AISafetyAtlas.LinearSystems.determinesStateOn_iff_isObservable", "AISafetyAtlas.LinearSystems.IsReachable", "AISafetyAtlas.LinearSystems.not_isReachable_of_not_isControllable" ], "module": "AISafetyAtlas.LinearSystems.Dynamics", "atlas_module": "AISafetyAtlas.LinearSystems.Dynamics", "build_command": "lake build AISafetyAtlas.LinearSystems.Dynamics", "scope_delta": { "summary": "ATLAS-ORIGINAL, added 2026-09-20, and the module that falsified this row's own central claim. Until it landed, every note on this row and on the survey rows BY-001 and BY-002 said that nothing in the tree defines a trajectory, a solution of the differential equation or an output signal, and that the reading 'the state cannot be reconstructed from the outputs' is therefore not proved. IsTrajectoryOn is x' = Ax + Bu asked on a SET of times rather than on the whole line, outputSignal is y = Cx, and DeterminesStateOn is complete state observability AS A PROPERTY - two trajectories under the same input agreeing on the output agree on the state - rather than as a criterion for it. determinesStateOn_iff_isObservable proves that property EQUIVALENT to the Kalman rank condition on any non-empty open window; the forward half needs Cayley-Hamilton, which the cluster had already spent in Hautus, and the converse needs only an eigenvector of the unobservable subspace, which turns the matrix exponential into a scalar one and is why no analysis layer was needed here. WIDER THAN PRINT: print writes the system on the whole line and these carry the window as a parameter. NOT VACUOUS: Examples.LinearSystems.blind_not_determinesStateOn is a system whose state its outputs do not determine and integrator_determinesStateOn_iff one where they do.", "evidence": "docs/provenance/source-coverage-audit.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "AISafetyAtlas.LinearSystems.flow", "AISafetyAtlas.LinearSystems.flow_mul_flow_neg", "AISafetyAtlas.LinearSystems.commute_flow", "AISafetyAtlas.LinearSystems.hasDerivAt_flow_entry", "AISafetyAtlas.LinearSystems.hasDerivAt_flow_mulVec", "AISafetyAtlas.LinearSystems.flow_isTrajectory", "AISafetyAtlas.LinearSystems.continuous_flow", "AISafetyAtlas.LinearSystems.drivingTerm", "AISafetyAtlas.LinearSystems.drivenState", "AISafetyAtlas.LinearSystems.drivenState_isTrajectory", "AISafetyAtlas.LinearSystems.adjointFlow", "AISafetyAtlas.LinearSystems.hasDerivAt_adjointFlow", "AISafetyAtlas.LinearSystems.mem_unobservableSubspace_of_adjointFlow_eq_zero", "AISafetyAtlas.LinearSystems.eq_zero_of_adjointFlow_eq_zero", "AISafetyAtlas.LinearSystems.adjointSignal", "AISafetyAtlas.LinearSystems.dotProduct_drivenState", "AISafetyAtlas.LinearSystems.eqOn_zero_of_intervalIntegral_eq_zero", "AISafetyAtlas.LinearSystems.adjointSignal_eq_zero_of_dotProduct_drivenState_eq_zero", "AISafetyAtlas.LinearSystems.reachedSet", "AISafetyAtlas.LinearSystems.exists_dotProduct_eq_zero_of_ne_top", "AISafetyAtlas.LinearSystems.IsCompletelyReachable", "AISafetyAtlas.LinearSystems.isCompletelyReachable_of_isControllable", "AISafetyAtlas.LinearSystems.isCompletelyReachable_iff_isControllable", "AISafetyAtlas.LinearSystems.isReachable_iff_isControllable" ], "module": "AISafetyAtlas.LinearSystems.Flow", "atlas_module": "AISafetyAtlas.LinearSystems.Flow", "build_command": "lake build AISafetyAtlas.LinearSystems.Flow", "scope_delta": { "summary": "ATLAS-ORIGINAL, added 2026-09-20, and the controllability half of the same retraction. flow A t is the matrix exponential e^(At), with its group law and its derivative taken from Mathlib's hasDerivAt_exp_smul_const under the scoped operator norm on matrices; drivenState is variation of constants, x(t) = Phi(t) * integral from 0 to t of Phi(-s) B u(s) ds, written with the two flow factors APART so that only the upper limit depends on the time, which turns the derivative into a product rule and a fundamental theorem of calculus instead of differentiation under the integral sign. isCompletelyReachable_iff_isControllable is print's property at print's own quantifier - a run between ANY two states, not only out of the origin - proved EQUIVALENT to the Kalman rank condition. THE GRAMIAN IS NOT USED AND WAS NOT NEEDED: reachedSet makes the states reached at a fixed positive time a subspace, exists_dotProduct_eq_zero_of_ne_top gets an annihilating covector through the DUAL rather than an inner product, dotProduct_drivenState reads that covector on the run as the adjoint signal paired with the input, and the input that exposes the silence is THE SIGNAL'S OWN CONJUGATE, so the integrand is a non-negative real. ONE MATHLIB GAP, assembled from two halves it does have: eqOn_zero_of_intervalIntegral_eq_zero, that a non-negative continuous function with vanishing interval integral is zero. NC-013 in docs/provenance/formalization-search.json records that no proof assistant held the continuous-time reachability equivalence when the six-corpus search was run on 2026-09-20. NO WINDOW PARAMETER on this side, and that asymmetry with Dynamics is disclosed in the section 25 Intro row: print writes the system with no domain restriction, so the whole line is print's own reading and there is no narrower class to fall short of. NOT VACUOUS: Examples.LinearSystems.deaf_not_isCompletelyReachable is a system that cannot be driven between arbitrary states, integrator_isCompletelyReachable one that can, and integrator_drivenState_eq_ramp shows the general construction reproduces the hand-written solution.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.LinearSystems.isObservable_iff_hautus", "type": "REFERENCE", "source_declarations": [ "isObservable_iff_hautus", "isControllable_iff_hautus", "isObservable_iff_observabilityMatrix_rank_eq", "isControllable_iff_controllabilityMatrix_rank_eq", "isControllable_iff_isObservable_transpose" ], "application": "Control: the standard algebraic tests for whether a linear time-invariant system's state is determined by its outputs and can be driven by its inputs, together with the differential system they are tests for. The criteria are adapted from AnandGokhale/LeanForControl; the dynamical layer is atlas-original. The trajectory bridge this field said was not in the tree was built on 2026-09-20 and this field swept 2026-09-21: determinesStateOn_iff_isObservable and isCompletelyReachable_iff_isControllable make each criterion equivalent to the property, so the dynamical reading is proved rather than owed." } ] }, "public": { "group": "Limits of control and regulation", "title": "Controllability and observability tests", "summary": "Two matrix tests that say whether a linear system's inputs can reach every state, and whether its outputs pin the state down - and, since 2026-09-20, the differential system they are tests for, with each test proved equivalent to the property it is named for.", "use": "When a linear model is on the table and the question is whether the state is determined by what is observed, or reachable by what can be driven - either as an algebraic test or as a statement about the runs themselves.", "attribution": "Kalman and Hautus; the criteria adapted from AnandGokhale/LeanForControl, the dynamical layer atlas-original" } }, { "id": "LAND-SOV-VOTINGPOWER-001", "name": "A zero power index names a null player only where the marginals cannot cancel", "tags": [ "multi-agent", "decision-theory" ], "notes": "GROUNDING. Source is the unpublished proposal pinned in docs/provenance/formal-power-proposal-triage.md; not in the coverage audit, not coverage. This row closes the proposal's Q10 and B6 together, because they are one theorem about one class and were both graded SUBSTRATE for the same missing object. A search on 2026-09-12 found Shapley in danlyng/Econlib four toolchain versions behind and found the Banzhaf index in NO Lean repository at all, so B6 was a build regardless and building the shared substrate made the Econlib port question moot for this purpose: Q10 needs the Shapley-Shubik FORMULA on simple games, not an axiomatic characterization, and nothing here claims one. CONTENT. SimpleGame carries val to the integers with a field saying only 0 and 1 occur; monotonicity is deliberately NOT a field, because both printed results hold only in the monotone class and a nonmonotone simple game has to remain an inhabitant of the same type for the countermodel to be statable. shapleyShubik uses the printed weights, banzhafRaw the uniform weight over the coalitions the player is not in. Both zero-value results are corollaries of one lemma, sum_weighted_eq_zero_iff: a positively weighted sum of nonnegative terms vanishes only termwise. shapleyWeight_pos supplies the weights, marginal_nonneg supplies the nonnegativity, and monotonicity is used for nothing else. B6's printed READING -- the score is the probability that changing the player's vote changes the outcome -- is banzhafRaw_eq_swings_div, which identifies the average with a count of pivotal coalitions over the number of coalitions, using that each marginal in this class is a 0/1 pivotality indicator. THE CLASS RESTRICTION IS NOT DECORATION. cancel is a simple nonmonotone game on two players whose two marginals are +1 and -1 with equal Shapley weights: the value is zero at a player who is not null. Without it, Q10's hypothesis would look like a formality. NOT VACUOUS. weighted is print's own example, quota 3 with weights (3, 1, 1): voter 1 carries a third of the nominal weight and is null, so the theorem is not a restatement of having some weight, while voter 0's value is nonzero. RAW IS NOT NORMALIZED. Only the raw Banzhaf score is defined. majority_banzhafRaw_sum_ne_one shows the raw scores of simple majority of three sum to 3/2, which is why a separate normalized index exists; print says the normalized index is a different normalization and Pitz and Ferraz, Cohesion-Sensitive Power Indices (arXiv 2026), page 7, state the classical value with exactly the weights used here and then normalize in their Definition 3.2. That paper is pinned privately, and is a NEIGHBOUR: nothing here is graded against it and it contributes no coverage.", "original_source_refs": [ "power-sovereignty-proposal-unpublished" ], "related_result_ids": [ "LAND-SOV-INFL-001", "LAND-SOV-EMPOWER-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "SimpleGame", "others", "SimpleGame.Monotone'", "SimpleGame.marginal", "SimpleGame.NullPlayer", "SimpleGame.Pivotal", "SimpleGame.shapleyWeight", "SimpleGame.shapleyShubik", "SimpleGame.banzhafRaw", "SimpleGame.sum_weighted_eq_zero_iff", "SimpleGame.marginal_nonneg", "SimpleGame.marginal_eq_one_iff_pivotal", "SimpleGame.nullPlayer_iff_others", "SimpleGame.shapleyWeight_pos", "SimpleGame.shapleyShubik_eq_zero_iff", "SimpleGame.banzhafRaw_eq_swings_div", "SimpleGame.banzhafRaw_eq_zero_iff", "SimpleGame.shapleyShubik_eq_zero_iff_banzhafRaw_eq_zero" ], "module": "AISafetyAtlas.Sovereignty.VotingPower", "build_command": "lake build AISafetyAtlas.Sovereignty.VotingPower", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished proposal; nothing graded against a pinned source. Against the proposal: Same for Q10's and B6's statements at the printed class, finite monotone simple games, with the printed weights. Wider in that the player type is an arbitrary Fintype with decidable equality rather than an initial segment, and in that the zero-value equivalences are stated as iffs where print argues one direction. Narrower and deliberately: no Shapley value in general, no axiomatic characterization, no efficiency and no linearity, since Q10 consumes only the formula on this class; and only the RAW Banzhaf score is defined, with the normalized index left out and its absence witnessed by majority_banzhafRaw_sum_ne_one rather than asserted. Print's remark that worst-case intervention influence loses the average-frequency information is NOT formalized here, because the atlas has no intervention-influence object at simple games to compare against.", "evidence": "docs/provenance/formal-power-proposal-triage.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.SimpleGame.shapleyShubik_eq_zero_iff", "type": "NEW_PROOF", "source_declarations": [ "SimpleGame", "SimpleGame.shapleyShubik", "SimpleGame.sum_weighted_eq_zero_iff" ], "application": "Auditing a governance arrangement by the votes it hands out: a zero power index certifies that a participant cannot change any decision, but only inside the class where marginals cannot cancel. Outside it a zero index is consistent with real influence, which cancel exhibits, so the certificate must carry its class." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.SimpleGame.banzhafRaw_eq_swings_div", "type": "NEW_PROOF", "source_declarations": [ "SimpleGame.banzhafRaw", "SimpleGame.Pivotal" ], "application": "Reading a power score as a frequency: the raw Banzhaf number is the fraction of the coalitions a participant might face in which that participant decides the outcome, so it answers how often rather than whether ever." } ] }, "public": { "group": "Limits of control and regulation", "title": "A vote with no power, and the fine print on measuring it", "summary": "In a voting rule where adding members never loses a vote, two standard power scores are zero for exactly the participants who can never change any outcome -- and a participant holding a third of the nominal weight can be one of them. Drop the monotonicity and a zero score stops meaning powerless, because gains and losses cancel.", "use": "When a share of votes, seats or weight is being offered as evidence of influence, or when a zero power index is being read as proof that a party is harmless.", "attribution": "Atlas-side interpretation of an unpublished proposal; the indices are Shapley-Shubik and Banzhaf, with the classical weights as stated by Pitz and Ferraz (2026), which is cited as a neighbour and not graded" } }, { "id": "LAND-SOV-CAPABILITY-001", "name": "A maintenance floor for fallback capability, and why assisted output does not report it", "tags": [ "control-theory", "agent-incentives" ], "result_shape": "ACHIEVABILITY", "notes": "GROUNDING. Source is the unpublished governance sketch catalogued as governance-kernel-sketch-unpublished, section 12.1, items GK1 and GK2; pinned and quoted in docs/provenance/capability-maintenance-floor.md. Not published, not in the coverage audit, not coverage. WHY IT EXISTS, NARROWLY. The verb preserve already has objects in this tree -- Sovereignty.RetainsFamily is preservation of what a coalition can force across a change of game form, and Wireheading.GoalPreservation is preservation of a goal across self-modification -- so the claim that it had none is wider than the tree supports and is not made. What had no object is CAPABILITY preservation over time: a scalar quantity that decays, is topped up each period, and must not fall below a floor. That is what this row adds. LAND-SOV-STEERING-001's note that the five verbs are named as interpretation and not formalized is about the verb inventory as such, not about the absence of preservation-shaped objects. CONTENT. AffineCapability is print's recurrence k(t+1) = a * k t + p t carried as a HYPOTHESIS on a supplied trajectory rather than as a defined function, because print calls it explicitly selected and says it is not an empirical universal law. maintenanceFloor_of_practice_floor is GK1: nonnegative retention, a start at or above the floor, and top-ups of at least (1 - a) * kmin keep the trajectory at or above kmin forever. It is stated at exactly print's hypotheses -- no a <= 1, no nonnegative skill, no nonnegative practice -- because print says in the next sentence that the theorem needs only its printed assumptions. Examples.Sovereignty.Capability.expanding_floor inhabits the antecedent with a = 2 so that claim is checked, and floor_fails_without_practice breaks the conclusion at a = 1 once the top-up floor is dropped, so both hypotheses are load-bearing rather than decorative. THE SECOND HALF. fallback_not_knowable_from_assistedOutput is GK2, as one application of Knowledge.not_knowable_of_collision: the fallback capability is not Knowable from assisted output, so the quantity the floor is about is not the quantity anyone observes. exists_output_rise_with_fallback_fall is the same failure in order-theoretic form, constructive at any two comparable levels, and Examples.Sovereignty.Capability.sketch_case is print's own arithmetic and no other numbers: fallback 2 to 1, borrowed 3, output 2 to 4. PRIOR ART, SEARCHED BEFORE BUILDING. Mathlib's Analysis.SpecificLimits.ArithmeticGeometric carries the same recurrence with a CONSTANT forcing term, and div_lt_arithGeom is its floor. It is not reused on four counts, each of which the model needs: the top-up varies with t and is precisely what an intervention moves, the bound is strict, a = 1 and a = 0 are excluded, and it requires a Field where this needs only an ordered ring. So this statement is a generalization of a Mathlib statement along four axes rather than an absence. THE THIRD HALF, BUILT 2026-09-13. GK3 -- adding optional strategies, with the old semantics simulated unchanged, cannot destroy an existing guarantee -- is monotonicity of Forces under enlargement of a coalition's strategy sets. PRIOR ART, FOUND AFTER THE BUILD AND RECORDED FOR THAT REASON: this claim is published. Wooldridge and van der Hoek 2005, Proposition 3 item 1, journal page 410, reads S |= <>phi -> <>phi for non-trivial normative systems with eta less restrictive than eta', and its proof on the same page is exactly this argument - by their Proposition 2, eta <= eta' gives Sigma-eta-prime-C contained in Sigma-eta-C, so the witnessing strategy survives, and comp(sigma_C, q) does not depend on eta because, in their own words at journal page 409, the semantics does not put any constraint on the agents outside coalition C. The search that should have found it was run against Lean and Isabelle corpora for GK3's shape and not against the deontic literature; the source was already held and unread in this repository at the time. The game-form embedding was costed as missing and is built here: Enlarges is a map from one GameForm to another that embeds every old option, requires that OUTSIDE the coalition every new option is an old one, and requires the old outcome on embedded profiles. Enlarges.forces is the claim and Enlarges.effectivity_subset is the same statement at the effectivity family. GRADED, so the reader need not reconstruct it: section 27 of docs/provenance/source-coverage-audit.md grades Enlarges.forces against Proposition 3 item 1 as Partial, wider in the carrier and NARROWER on the temporal axis, and that narrowing is costed there rather than closed. ONE CLAUSE CARRIES IT, AND IT IS THE ONE ABOUT EVERYBODY ELSE: Examples.Sovereignty.Capability.foe_breaks_forces keeps the embedding and the unchanged semantics, drops only the requirement that the complement gained nothing, and loses the guarantee -- so the unqualified reading of GK3, more options cannot hurt, is false and the qualified one is a theorem. wide_forces_one and base_not_forces_one show the enlargement in the positive witness is a real one and not an identity. This does NOT discharge the sketch's fallback-effectivity object; see the last paragraph. NOT HERE. Section 12.2's hazard bound needs a probability layer and is not in this row. AND THE SKETCH'S OWN WIDER OBJECT IS NOT THIS ONE: the sentence after GK1 says the operational object is not skill alone but fallback effectivity, what the institution can guarantee when a specified dependency is withdrawn. This row carries the scalar recurrence GK1 is stated over; the effectivity reading is not built.", "original_source_refs": [ "governance-kernel-sketch-unpublished", "wooldridge-van-der-hoek-2005-obligations-normative-ability" ], "related_result_ids": [ "LAND-KNOW-001", "LAND-KNOW-UNIFORM-001", "LAND-SOV-STEERING-001", "LAND-SOV-POWER-001", "LAND-SOV-DEONTIC-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Sovereignty.AffineCapability", "AISafetyAtlas.Sovereignty.MaintenanceFloor", "AISafetyAtlas.Sovereignty.maintenanceFloor_of_practice_floor", "AISafetyAtlas.Sovereignty.MaintenanceFloor.le_zero", "AISafetyAtlas.Sovereignty.assistedOutput", "AISafetyAtlas.Sovereignty.fallbackCapability", "AISafetyAtlas.Sovereignty.fallback_not_knowable_from_assistedOutput", "AISafetyAtlas.Sovereignty.exists_output_rise_with_fallback_fall", "AISafetyAtlas.Sovereignty.Enlarges", "AISafetyAtlas.Sovereignty.Enlarges.forces", "AISafetyAtlas.Sovereignty.Enlarges.effectivity_subset" ], "module": "AISafetyAtlas.Sovereignty.Capability", "atlas_module": "AISafetyAtlas.Sovereignty.Capability", "build_command": "lake build AISafetyAtlas.Sovereignty.Capability", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished sketch; nothing graded against a pinned published source. Against the sketch's section 12.1: Same for GK1's statement, at exactly its printed hypotheses, and Same for GK2's reference case, at exactly its printed numbers. WIDER in the carrier: print writes a real-valued skill, and this holds in any ring with a compatible partial order, commutativity not required. WIDER in what GK2 becomes: print exhibits one numeric pair, and this proves the collision at any two distinct capability levels and the order failure at any two comparable ones. NARROWER in nothing. Two declarations have no counterpart in print: MaintenanceFloor.le_zero, which records that the start hypothesis is not decoration, and fallback_not_knowable_from_assistedOutput, which routes GK2 through this repository's knowability kernel rather than leaving it as arithmetic. Against Mathlib's div_lt_arithGeom, which is the nearest existing statement: WIDER on four axes -- varying rather than constant top-up, non-strict rather than strict bound, a = 0 and a = 1 admitted, and an ordered ring rather than a field. Against the sketch's GK3: Same in the claim. AGAINST THE PUBLISHED FORM, Wooldridge and van der Hoek 2005 Proposition 3 item 1, journal page 410: WIDER in the carrier - print fixes one AATS and varies only which strategies a normative system permits, so its embed is a set inclusion and its outcome map never moves, while Enlarges takes two arbitrary game forms over arbitrary agent and outcome types with an arbitrary embedding and an outcome map that may differ off the image. NARROWER on the temporal axis: print's conclusion is about path formulae over infinite computations and this is the one-shot fragment, with a set of outcomes where print has a set of paths and no temporal operators at all. That pair is graded Partial in section 27 of docs/provenance/source-coverage-audit.md, and the temporal axis is costed there - closing it means giving the Sovereignty cluster paths, which is the same cost that section's §2 row names and which no single row owns. The counterexample foe_breaks_forces has no counterpart in print, and cannot: print's complement is unconstrained by construction, so the hypothesis it isolates is a fact about print's model rather than a clause print states. No complexity claim, no rate claim, and no claim that the recurrence describes any real capability: print calls it explicitly selected and this row carries it as a hypothesis for that reason.", "evidence": "docs/provenance/capability-maintenance-floor.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.maintenanceFloor_of_practice_floor", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Sovereignty.AffineCapability", "AISafetyAtlas.Sovereignty.MaintenanceFloor" ], "application": "Sizing a practice or drill requirement. If a capability decays at a known rate and must not fall below a stated floor, this says how large each top-up has to be, and says it as a per-step condition an operator can check rather than as a limit." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.fallback_not_knowable_from_assistedOutput", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Knowledge.not_knowable_of_collision" ], "application": "Refusing assisted throughput as evidence of retained capability. A team whose output is measured with the assistance in place has produced no reading at all of what it could do without it, and this is the reason stated as an impossibility rather than as a caution." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.exists_output_rise_with_fallback_fall", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Sovereignty.assistedOutput" ], "application": "Reading a rising performance metric during a dependency rollout. Improvement in the observed number is compatible with strict decline in what survives withdrawal of the dependency, so the metric cannot be used to license further dependence." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.Enlarges.forces", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Sovereignty.Forces" ], "application": "Deciding whether a new capability may be added to a system that already carries a guarantee. Adding options to the party that holds the guarantee, while leaving the old behaviour available and meaning the same thing, cannot cost the guarantee; adding options to anybody else can, and the accompanying witness shows it does." } ] }, "public": { "group": "Limits of control and regulation", "title": "A skill that is topped up enough never falls below its floor -- and the number you can see does not tell you whether it did", "summary": "If a capability fades at a fixed rate and is refreshed by at least a computable amount each period, it stays above a chosen minimum forever. Separately, the output a team produces while being assisted cannot reveal how much of that output would survive if the assistance were withdrawn: output can rise while the surviving capability falls.", "use": "When deciding how much practice a dependency must be paired with, and when someone offers assisted throughput as evidence that capability has been retained.", "attribution": "Atlas-side interpretation of an unpublished sketch; the recurrence is a selected model and is carried as a hypothesis, not asserted" } }, { "id": "LAND-EVAL-BLINDSPOT-001", "name": "What an evaluation that scores runs one at a time can and cannot see", "tags": [ "compositionality", "oversight", "verification" ], "result_shape": "CHARACTERIZATION", "notes": "BRIDGE. Layer 2 is AISafetyAtlas.Compositional.Hyperproperties, which is about trace systems and sets of them and names no evaluation. This row is the arrow to Reuel, Bucknall et al. Open Problems 18 ('How can the thoroughness of evaluations be measured?') and 19 ('How can potential blind spots of evaluations be identified?'). CONTENT. An evaluation is modelled as a PER-RUN SCORE: it executes runs, records something about each, and its whole evidence about the system is the set of scores seen. traceProperty_knowable_of_score_decides is the positive half -- any requirement of the form 'every run is acceptable', a trace property, is settled exactly by such an evaluation when the score carries acceptability. sampling_misses_subsingleton is the negative half: the moment the score maps two distinct runs to the same value, which is what recording a score rather than the run means, a property of the run SET escapes, and no number of further runs helps, because the two systems confused produce identical evidence at any length and any repetition. BOTH HALVES ARE STATED because either alone misleads: the first is why evaluation suites work, the second is why thoroughness-as-coverage is the wrong instrument outside trace properties. So Open Problem 18 is answered conditionally and Open Problem 19 is answered structurally -- the blind spot is not a sampling budget. WHAT IS NOT CLAIMED. That any particular benchmark is per-run scoring is a claim about that benchmark and is layer 4; no evaluation suite is surveyed here. An evaluation that compares runs to each other, recording pairs or the whole batch, is not modelled by `score` at all. Set.Subsingleton is used as the witness hyperproperty because it is the smallest one that is not a trace property, and is not intended as a specific safety requirement. GROUNDING. Examples.Compositional.Hyperproperties.Evaluation inhabits both halves at the SAME evaluation -- a pass/fail record over three behaviours where two distinct runs both fail and the record does not say which -- so neither half is a conditional nothing satisfies. An evaluation that recorded which run failed would be injective there and would not collide, which is the honest boundary of the obstruction. SOURCE STATUS. The TMLR paper is a DIRECTORY: it states open problems and no theorem, so this row claims no coverage of it and it is not in the coverage audit.", "original_source_refs": [ "reuel-bucknall-2025-open-problems-technical-ai-governance" ], "root_import": true, "public": { "group": "Limits of verification", "title": "Some things cannot be checked one run at a time", "summary": "An evaluation that scores every run a system can produce, with a score that decides acceptability, settles every requirement of the form 'each run must be acceptable'. It cannot settle requirements about the collection of runs taken together -- such as 'the system behaves in only one way' -- as soon as it records a summary rather than the whole run, because two different systems then produce exactly the same record. Running it more times does not help: the missing information was never in the record.", "use": "When an evaluation result is offered as evidence that a system is safe, and when deciding whether a gap in an evaluation is answered by more testing or needs a different instrument.", "attribution": "Atlas-side bridge: the mathematics is this repository's hyperproperty layer; the questions are Open Problems 18 and 19 of Reuel, Bucknall et al., TMLR 04/2025, which the theorems bound and do not answer" }, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Compositional.Hyperproperties.Evaluation.scoreSet", "AISafetyAtlas.Compositional.Hyperproperties.Evaluation.traceProperty_knowable_of_score_decides", "AISafetyAtlas.Compositional.Hyperproperties.Evaluation.not_knowable_of_score_collision", "AISafetyAtlas.Compositional.Hyperproperties.Evaluation.sampling_misses_subsingleton" ], "module": "AISafetyAtlas.Compositional.Hyperproperties.Evaluation", "atlas_module": "AISafetyAtlas.Compositional.Hyperproperties.Evaluation", "build_command": "lake build AISafetyAtlas.Compositional.Hyperproperties.Evaluation AISafetyAtlas.Examples.Compositional.Hyperproperties.Evaluation" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Compositional.Hyperproperties.Evaluation.traceProperty_knowable_of_score_decides", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.Knowable" ], "application": "Justifying an evaluation suite. A requirement that every run must be acceptable is settled by scoring runs when every possible run is scored and the score decides acceptability; for that class of requirement full coverage is the right measure of thoroughness. A finite passing suite scores only some runs, and nothing here is about sampled evaluations.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-traceproperty-knowable-of-score-decides.md" } }, { "atlas_declaration": "AISafetyAtlas.Compositional.Hyperproperties.Evaluation.sampling_misses_subsingleton", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.not_knowable_of_collision" ], "application": "Deciding whether a gap in an evaluation is answered by more testing. If the evaluation records a summary of each run rather than the run, some requirement about the set of runs is invisible to it at every sampling budget, so the response is a different instrument and not a longer run.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-sampling-misses-subsingleton.md" } } ] } }, { "id": "LAND-AUDIT-REGISTRY-001", "name": "Audit registries along a value chain: what publishing declarations can settle", "tags": [ "information-theory", "oversight", "multi-agent" ], "result_shape": "CHARACTERIZATION", "notes": "BRIDGE. Layer 2 is AISafetyAtlas.Oversight.JointObservation, which is about principals, evidence and coverage and names no audit scheme. This row is the arrow to Reuel, Bucknall et al. Open Problem 60 ('How can audit registries be used to provide end-to-end verification along the AI value chain?') and Open Problem 101 (an auditable log of all actors). CONTENT. EvidenceArchitecture already separates what a principal HOLDS (privateState) from what it DECLARES (emit), and the module already had the singleton pair -- localCandidate reads one principal's declaration, privateSingletonCandidate reads its evidence. What was missing is the COALITION level, which is the level the question is asked at. registryCandidate is every member's filing; consortiumCandidate is the same actors pooling the evidence itself. consortium_covers_of_registry_covers holds with no hypotheses at every coalition: a registry is never stronger than the evidence behind it. not_registry_covers_of_emit_collision is the obstruction, and it is quantified over an ARBITRARY coalition, which is the sharp part: two executions on which every member files identically and the hazard differs refute coverage however many actors the chain contains. Enrolling more filers repairs it only if some new filing separates the two executions; filers whose filings also agree cannot, because the obstruction is in what a filing discards rather than in how many were gathered -- so end-to-end is the wrong axis when the failure is per-actor, and the repair is a finer declared interface. WHAT IS NOT CLAIMED. Value-chain actors are not modelled; who occupies which position in a real pipeline is a layer-4 assignment and is not made. Nothing says a real registry schema loses what it needs: emit is an arbitrary projection and may be injective, in which case the collision does not arise and both candidates cover exactly the same hazards. Coverage is informational and under truthful reporting, as CandidateObservation.observe fixes; a registry whose filers may misreport is a different question this module does not touch. GROUNDING. Examples.Oversight.JointObservationRegistry inhabits both halves on a supplier and a deployer holding exact component versions and filing version BANDS. The filings are genuine information, not vacuous -- the witness turns on the band being coarser than the hazard needs, with versions (0,0) and (0,1) filing identically while one pair matches and the other does not. SOURCE STATUS. The TMLR paper is a DIRECTORY: it states open problems and no theorem, so this row claims no coverage of it and it is not in the coverage audit.", "original_source_refs": [ "reuel-bucknall-2025-open-problems-technical-ai-governance" ], "root_import": true, "public": { "group": "What an observer can recover", "title": "A filing is only as good as what it leaves out", "summary": "When every party in a supply chain files a declaration, the declarations together can settle any question the underlying evidence settles -- but only if each declaration keeps what the question needs. If two different situations produce identical filings from everyone, no registry built from those filings can tell them apart, and adding more parties to the chain helps only if some new party's filing tells them apart; parties whose filings also agree add nothing. The fix is a more informative filing, not a longer chain.", "use": "When an audit registry or disclosure regime is proposed as end-to-end assurance, and when deciding whether a gap is answered by enrolling more parties or by changing what each party must disclose.", "attribution": "Atlas-side bridge: the mathematics is this repository's joint-observation layer; the question is Open Problem 60 of Reuel, Bucknall et al., TMLR 04/2025, which the theorems bound and do not answer" }, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Oversight.JointObservation.registryCandidate", "AISafetyAtlas.Oversight.JointObservation.consortiumCandidate", "AISafetyAtlas.Oversight.JointObservation.consortium_covers_of_registry_covers", "AISafetyAtlas.Oversight.JointObservation.not_registry_covers_of_emit_collision", "AISafetyAtlas.Oversight.JointObservation.consortium_covers_and_registry_does_not" ], "module": "AISafetyAtlas.Oversight.JointObservation.Registry", "atlas_module": "AISafetyAtlas.Oversight.JointObservation.Registry", "build_command": "lake build AISafetyAtlas.Oversight.JointObservation.Registry AISafetyAtlas.Examples.Oversight.JointObservationRegistry" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Oversight.JointObservation.consortium_covers_of_registry_covers", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Oversight.JointObservation.Covers" ], "application": "Justifying a disclosure regime. Any question a set of filings settles is settled by the evidence those filings were computed from, at every coalition and with no side condition, so a registry never settles more than the evidence behind it; this bounds a registry from above and says nothing about whether filing is worth requiring.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-03", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-consortium-covers-of-registry-covers.md" } }, { "atlas_declaration": "AISafetyAtlas.Oversight.JointObservation.not_registry_covers_of_emit_collision", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.not_knowable_of_collision" ], "application": "Diagnosing a registry that fails to establish something. If two situations produce identical filings from every party, the failure is the disclosure schema and not the coverage of the chain, and enrolling further parties whose filings also agree on the two situations cannot repair it; only a filing that separates them can.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-03", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-not-registry-covers-of-emit-collision.md" } } ] } }, { "id": "LAND-ACCESS-ORDER-001", "name": "Forms of model access: three points on the informativeness order, and what no methodology repairs", "tags": [ "information-theory", "oversight", "verification" ], "result_shape": "CHARACTERIZATION", "notes": "BRIDGE. Layer 2 is AISafetyAtlas.Knowledge, which is about observation maps and decoders and names no model, no weights and no API. This row is the arrow to Reuel, Bucknall et al. Open Problem 37 ('What research and auditing methodologies are possible given a range of forms of access on the continuum between black- and white-box access?'). CONTENT. Determines is already a preorder on observation maps, so the continuum has a home; what the kernel cannot supply is points on it. AccessSetting models an artifact with two things a run exposes -- an output and a richer per-input score -- and the three access levels are three observations of the SAME weights: whiteBox (the artifact), scoreAccess (the score function), blackBox (the input-output behaviour). whiteBox_determines_scoreAccess and scoreAccess_determines_blackBox place them in the order and whiteBox_determines_blackBox composes them, so Knowable.mono is exactly R#37's monotonicity: what a weaker access level establishes, a stronger one establishes. whiteBox_knowable is the top and takes no hypothesis, because reading the artifact is the identity. The sharp half is no_blackBox_methodology: if a property does not follow from behaviour then NO methodology over behaviour establishes it -- analysis is an arbitrary function of the entire input-output behaviour, so it may probe adaptively, aggregate over unboundedly many queries or run any statistic, and it is still a function of evidence that was already insufficient. R#37 asks which methodologies an access level permits; the answer is a property of the access level and not of the method. exists_indistinguishable_behaviour makes a negative answer actionable by handing the auditor the two artifacts the access cannot separate. WHAT IS NOT CLAIMED. No real access level is placed in the order. whiteBox, scoreAccess and blackBox are three projections of THIS structure; whether a deployment's API is behaviour, whether returned logprobs are scores, and where fine-tuning access or activation reads sit are layer-4 assignments and none is made. The order is proved between the projections, not between the products. readOut_scores is the one modelling commitment -- the output is recoverable from the score, which is deterministic decoding; under sampling the output-level observation is a distribution and the chain is not stated there. Knowable is EXACT recovery, so statistical and approximate auditing is not covered. Knowable is also the EXISTENCE OF A DECODER AND NOT ITS COMPUTABILITY: whiteBox_knowable must not be read as 'verification is possible with full access', since a property can be knowable from the artifact in this sense and have no algorithm deciding it -- AISafetyAtlas.Computability.rice_code_iff says a decision procedure over the code exists only in the two trivial cases for an extensional property, and the code is full access. The order measures what the evidence CONTAINS, which is prior to and does not imply what can be computed from it. Reuel et al. R#38 and R#39 (how access affects misuse risk and model-theft risk) are NOT addressed: those are about what an adversary can DO with access, which is a capability and not a decoder-existence question. GROUNDING. Examples.Knowledge.Access exhibits a chain strict at BOTH steps, which is what keeps the conditionals from being vacuous. An artifact is three bits -- answer, confidence, and a dormant bit no input reaches -- over a single input, so nothing turns on running the system more. Confidence separates scoreAccess from blackBox; the dormant bit separates whiteBox from scoreAccess, and its two artifacts are behaviourally and score-wise identical at every input, so no evaluation distinguishes them and no analysis of the score record does either. SOURCE STATUS. The TMLR paper is a DIRECTORY: it states open problems and no theorem, so this row claims no coverage of it and it is not in the coverage audit.", "original_source_refs": [ "reuel-bucknall-2025-open-problems-technical-ai-governance" ], "root_import": true, "public": { "group": "What an observer can recover", "title": "More access answers more questions, and no cleverness substitutes for it", "summary": "Ways of examining an AI system can be ranked by how much they reveal: reading the system itself reveals at least as much as seeing its scores, which reveals at least as much as seeing only its answers. Anything establishable from a weaker form of examination is establishable from a stronger one. The other direction is the useful part: if a question cannot be answered from what a form of access reveals, then no analysis of that evidence answers it either -- not more queries, not adaptive probing, not any statistic -- because the missing information was never collected. A worked case shows all three levels genuinely differ, with a property that no amount of testing can detect and that reading the system settles at once.", "use": "When deciding what level of access an audit or evaluation requires, and when someone proposes to substitute a more sophisticated testing methodology for a higher level of access.", "attribution": "Atlas-side bridge: the mathematics is this repository's knowability kernel; the question is Open Problem 37 of Reuel, Bucknall et al., TMLR 04/2025, which the theorems answer inside a model of access and do not answer for any named real access level" }, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.Access.AccessSetting", "AISafetyAtlas.Knowledge.Access.whiteBox", "AISafetyAtlas.Knowledge.Access.scoreAccess", "AISafetyAtlas.Knowledge.Access.blackBox", "AISafetyAtlas.Knowledge.Access.whiteBox_determines_blackBox", "AISafetyAtlas.Knowledge.Access.whiteBox_knowable", "AISafetyAtlas.Knowledge.Access.no_blackBox_methodology", "AISafetyAtlas.Knowledge.Access.exists_indistinguishable_behaviour" ], "module": "AISafetyAtlas.Knowledge.Access", "atlas_module": "AISafetyAtlas.Knowledge.Access", "build_command": "lake build AISafetyAtlas.Knowledge.Access AISafetyAtlas.Examples.Knowledge.Access" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.Access.whiteBox_determines_blackBox", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.Determines.trans" ], "application": "Ranking forms of examination. Three access levels over one modelled artifact form a chain in the informativeness order, so anything establishable from a weaker level is establishable from a stronger one. Which real access level sits where is not claimed.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-whitebox-determines-blackbox.md" } }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Access.no_blackBox_methodology", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.not_knowable_comp" ], "application": "Refusing a methodology substitute for access. If a property does not follow from behaviour, no analysis of behaviour establishes it, however adaptive or however many queries it uses -- so a proposal to answer the question with a better test rather than more access can be rejected on the evidence, not on judgement.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-no-blackbox-methodology.md" } }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Access.exists_indistinguishable_behaviour", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.exists_witness_of_not_knowable" ], "application": "Reporting a negative audit result. An auditor who cannot settle a question from a given access level can exhibit the two systems that access cannot tell apart, which turns 'we could not determine this' into an inspectable obstruction.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-exists-indistinguishable-behaviour.md" } } ] } }, { "id": "LAND-GOODHART-REGTARGET-001", "name": "Regulatory targets: the bar certifies exactly the systems its evidence never covered", "tags": [ "agent-incentives", "decision-theory", "oversight" ], "result_shape": "CHARACTERIZATION", "notes": "BRIDGE. Layer 2 is AISafetyAtlas.Goodhart.Extremal, which is about a proxy M, a threshold c and a relationship fitted on a region R, and names no regulator and no rule. This row is the arrow to Reuel, Bucknall et al. Open Problem 92 ('What system properties (if any) are the most reliable indicators of risk, and thus candidates for serving as regulatory targets?'). R#92 has two halves and only the second is addressed: WHICH property indicates risk is empirical; what happens to a property ONCE IT IS MADE A TARGET is what the theorems state, and the phrase 'and thus candidates for serving as regulatory targets' is exactly the step. CONTENT. RegulatoryScheme is a bright-line rule: an indicator on systems, a threshold, and the EVIDENCE BASE -- the systems on which the indicator's link to risk was established. The one structural hypothesis is observedCeiling_lt_threshold, which puts the scheme in the EXTREMAL REGIME: the bar demands more than any studied system displayed. That is a modelling choice and not a definition of a rule -- a threshold set inside the studied range certifies some systems that were studied, certified_disjoint_evidenceBase does not apply to it, and that regime is not modelled here and is not claimed to be sound either. certified_disjoint_evidenceBase and certified_systems_were_never_examined are then immediate -- every system the rule certifies is one the rule's evidence never covered. risk_unconstrained_on_certified says what that costs, and the quantifier is the content: for ANY risk fitting the link on the evidence base and ANY displacement d there is a second risk fitting that link equally well, agreeing with the first across the whole evidence base, and differing by exactly d at every certified system. Not 'two models disagree somewhere', which any two functions satisfy. raising_the_bar_does_not_help and risk_unconstrained_at_every_higher_bar are the policy-legible corollary: tightening the threshold moves the certified set FURTHER from the evidence and the underdetermination holds at every higher bar, so the instrument that looks like the remedy is the same instrument. The positive half is stated first among the consequences, deliberately: risk_bounded_on_evidence_base says that INSIDE the evidence base the indicator predicts risk to the tolerance the fit was established at, which is why measuring indicators is worth doing and is what the rest of the module does not deny. WHAT IS NOT CLAIMED. No proposed regulatory target is evaluated; R#92's first half is empirical and nothing here bears on it. The regulated party's behaviour is NOT modelled -- there is no optimizer, no deadline and no enforcement, and selectedAt is a set of systems clearing a number, which is what a bright line selects and not how a firm responds to one. Nothing is quantified: d is arbitrary, so the underdetermination is total rather than measured, and no rate and no ranking of proxies follows. GROUNDING. Examples.Goodhart.RegulatoryTarget inhabits the structural hypothesis at a link fitted EXACTLY (epsilon = 0, risk equals the indicator) on systems scoring at most 1, with the bar at 2. The exact fit is chosen so the gap cannot be read as a sloppy measurement: the fit is perfect where it was taken. 3 is certified and concretely outside the evidence base, and the displacement is arbitrary at a bar of 100 as much as at 2. SOURCE STATUS. Two sources, neither carried as coverage. The TMLR paper is a DIRECTORY: it states open problems and no theorem. Manheim and Garrabrant arXiv:1803.04585v4 is cited by AISafetyAtlas.Goodhart.Extremal for its MODEL and numbers no theorem, no proposition, no lemma and no corollary, so this bridge rests on this repository's atlas-original sharpening of that prose and not on a published theorem's authority.", "original_source_refs": [ "reuel-bucknall-2025-open-problems-technical-ai-governance" ], "root_import": true, "public": { "group": "Limits of control and regulation", "title": "A threshold certifies the cases nobody studied", "summary": "Suppose a measurable property of a system is picked as a regulatory target because it tracks risk, and a threshold is set that a system must clear. Suppose further that the threshold is set above anything the studied systems displayed, which is the case where a rule is meant to bind on systems more capable than those anyone has examined. Then every system it certifies falls outside the population where the property's link to risk was actually established. Inside that population the property predicts risk as well as it ever did. Outside it, the same evidence is consistent with the true risk being off by any amount at all, at every certified system. Raising the threshold does not repair this: every higher bar also certifies only systems outside the evidence base.", "use": "When a measurable property is proposed as a compliance threshold, and when the response to a rule that seems too weak is to demand a higher score rather than to widen the evidence base.", "attribution": "Atlas-side bridge: the mathematics is this repository's extremal-Goodhart layer, itself an atlas-original sharpening of Manheim and Garrabrant (arXiv:1803.04585v4), which numbers no theorem; the question is Open Problem 92 of Reuel, Bucknall et al., TMLR 04/2025, whose first half -- which property indicates risk -- is empirical and is not addressed" }, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Goodhart.RegulatoryTarget.RegulatoryScheme", "AISafetyAtlas.Goodhart.RegulatoryTarget.certified", "AISafetyAtlas.Goodhart.RegulatoryTarget.risk_bounded_on_evidence_base", "AISafetyAtlas.Goodhart.RegulatoryTarget.certified_disjoint_evidenceBase", "AISafetyAtlas.Goodhart.RegulatoryTarget.certified_systems_were_never_examined", "AISafetyAtlas.Goodhart.RegulatoryTarget.risk_unconstrained_on_certified", "AISafetyAtlas.Goodhart.RegulatoryTarget.raising_the_bar_does_not_help", "AISafetyAtlas.Goodhart.RegulatoryTarget.indicator_informative_and_certification_unconstrained" ], "module": "AISafetyAtlas.Goodhart.RegulatoryTarget", "atlas_module": "AISafetyAtlas.Goodhart.RegulatoryTarget", "build_command": "lake build AISafetyAtlas.Goodhart.RegulatoryTarget AISafetyAtlas.Examples.Goodhart.RegulatoryTarget" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Goodhart.RegulatoryTarget.certified_systems_were_never_examined", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Goodhart.Extremal.selected_disjoint_observed" ], "application": "Reading a compliance threshold. Because the bar is assumed (a field of the scheme) to sit above what the evidence base exhibits, no certified system lies inside that evidence base -- so the population a rule certifies and the population its justification was built on are disjoint by construction.", "review_status": "STATEMENT_REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": false, "evidence": "docs/interpretation-reviews/review-certified-systems-were-never-examined.md" } }, { "atlas_declaration": "AISafetyAtlas.Goodhart.RegulatoryTarget.risk_unconstrained_on_certified", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Goodhart.Extremal.fits_underdetermined_off_observed" ], "application": "Bounding what a certification licenses. Whatever the true risk is, the evidence behind the indicator permits it to be off by any amount named, uniformly across every certified system -- so a certification carries no inference about risk at the systems it certifies, however good the fit was where it was taken.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-risk-unconstrained-on-certified.md" } }, { "atlas_declaration": "AISafetyAtlas.Goodhart.RegulatoryTarget.raising_the_bar_does_not_help", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Goodhart.Extremal.selected_disjoint_observed" ], "application": "Rejecting the obvious remedy. Tightening a threshold moves the certified set further from the evidence base rather than closer, and the underdetermination holds at every higher bar, so demanding a higher score does not repair a rule whose evidence does not reach the systems it certifies. A repair would need evidence reaching the certified systems; in this model the bar is assumed to sit above the evidence base, so that repair is not modelled here.", "review_status": "STATEMENT_REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": false, "evidence": "docs/interpretation-reviews/review-raising-the-bar-does-not-help.md" } } ] } }, { "id": "LAND-SOV-ASSESSMENT-001", "name": "Measuring unaided capability: withdrawal testing is forced, not chosen", "tags": [ "information-theory", "oversight" ], "result_shape": "CHARACTERIZATION", "notes": "BRIDGE. Layer 2 is AISafetyAtlas.Sovereignty.Capability, which holds the sum decomposition assistedOutput = fallback + borrowed and the fact that a sum does not determine its summands. This row is the arrow to a MEASUREMENT question about cognitive offloading, stated over the two protocols actually used to answer it. SOURCE. International AI Safety Report 2026 (chair Yoshua Bengio; expert panel nominated by 30+ countries), section 2.3.2 'Risks to human autonomy': 'one study found that three months after the introduction of AI support, clinicians' ability to detect tumours WITHOUT AI ASSISTANCE had dropped by 6%'. The report's own caveat travels with every use of it: 'research into the relationship between use of AI and cognitive offloading and critical thinking is nascent, and further studies supporting these findings are warranted.' CONTENT. 'Ability without assistance' IS fallbackCapability, so the question is how anyone would ever measure it. Two protocols observe the same state. outputProtocol reads the assisted output, which is what an observer of ordinary work has. withdrawalProtocol reads the fallback directly, by taking the assistance away and testing. withdrawal_recovers_fallback is the positive half and is immediate -- the protocol observes the quantity, which is exactly what makes it expensive. no_procedure_on_output_recovers_fallback is the negative half and is why the positive one matters: not only does assisted output fail to determine fallback, NO FUNCTION of assisted output does, so any statistic, rubric, satisfaction score or productivity metric computed from delivered work inherits the failure. That is not_knowable_comp, and it makes the obstruction structural rather than a matter of instrument quality. protocols_are_incomparable is the sharp form: NEITHER protocol is more informative than the other, so withdrawal testing is not 'more of' output observation and no quantity of observed work approaches it, while withdrawal testing in turn says nothing about assisted performance. THEREFORE the 6% study is not an instance of the obstruction; it is the EXPENSIVE WAY AROUND IT, and the theorems say why withdrawal testing is the measurement the obstruction forces rather than a costly option a cheaper design could replace. WHAT IS NOT CLAIMED. That any deployed system causes cognitive offloading -- nothing here is evidence of decline, and the report's nascent-research caveat is part of the claim and not a footnote. That self-report is covered: no_procedure_on_output_recovers_fallback is a CONDITIONAL applying to a report IF that report is a function of assisted output, and whether a person's self-assessment is such a function is a layer-4 premise -- plausible, since someone judging their own competence largely sees their own finished work, and NOT ESTABLISHED HERE. The empirical literature describes a BIAS mechanism; this module describes NON-INVERTIBILITY; they are compatible, they are not the same claim, and neither follows from the other. That decline is affine: AffineCapability and maintenanceFloor_of_practice_floor sit in the same layer-2 module and are DELIBERATELY NOT USED, because that recurrence is in its own docstring 'a selected model, not a law' -- no measured human capability is claimed to satisfy it and nothing here predicts, explains or corroborates the 6% figure. That any practice policy, quantity or schedule follows. The additive decomposition is itself a model; a setting where assistance and capability do not compose additively is not this one. GROUNDING. Examples.Sovereignty.CapabilityAssessment fixes ZZ, because a Ring can be trivial and then x != y is uninhabited and every conditional says nothing. competent = (10, 0) and dependent = (4, 6) produce the SAME finished work 10 and differ by more than half in unaided capability. sophistication_does_not_help runs the identity as the rubric -- unbounded range, still blind -- so the obstruction is in the evidence and not in the instrument's resolution, and output_can_rise_while_capability_falls shows the observable can read the trend BACKWARDS. SOURCE STATUS. The report is a DIRECTORY: it surveys evidence and states no theorem, so this row claims no coverage of it and it is not in the coverage audit.", "original_source_refs": [ "bengio-2026-international-ai-safety-report" ], "root_import": true, "public": { "group": "What an observer can recover", "title": "You cannot read unaided skill off assisted work", "summary": "Suppose what a person produces with a tool is their own ability plus whatever the tool contributed. Then watching the finished work tells you nothing reliable about the ability on its own -- and this is not a matter of better metrics: any score, rubric or productivity measure computed from that work reads the same evidence and inherits the same blindness. Two situations can produce identical output while the underlying ability differs sharply, and the output can even rise while the ability falls. Taking the assistance away and testing does settle the question, and it is not a more thorough version of watching the work -- it is a different measurement that nothing else approaches.", "use": "When deciding whether a deskilling or dependence concern can be monitored through ordinary performance data, and when a withdrawal-based assessment is proposed and its cost questioned.", "attribution": "Atlas-side bridge: the mathematics is this repository's capability-decomposition layer; the measurement question is from the International AI Safety Report 2026 section 2.3.2, whose own caveat is that this research is nascent and further supporting studies are warranted. Nothing here is evidence that any decline occurred, and no model of how capability changes over time is used" }, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Sovereignty.CapabilityAssessment.outputProtocol", "AISafetyAtlas.Sovereignty.CapabilityAssessment.withdrawalProtocol", "AISafetyAtlas.Sovereignty.CapabilityAssessment.withdrawal_recovers_fallback", "AISafetyAtlas.Sovereignty.CapabilityAssessment.no_procedure_on_output_recovers_fallback", "AISafetyAtlas.Sovereignty.CapabilityAssessment.protocols_are_incomparable", "AISafetyAtlas.Sovereignty.CapabilityAssessment.withdrawal_settles_and_no_output_procedure_does" ], "module": "AISafetyAtlas.Sovereignty.CapabilityAssessment", "atlas_module": "AISafetyAtlas.Sovereignty.CapabilityAssessment", "build_command": "lake build AISafetyAtlas.Sovereignty.CapabilityAssessment AISafetyAtlas.Examples.Sovereignty.CapabilityAssessment" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.CapabilityAssessment.no_procedure_on_output_recovers_fallback", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.not_knowable_comp" ], "application": "Rejecting output-based monitoring of dependence. If a capability is observed only through the sum of that capability and borrowed assistance, no score computed from the observation recovers it -- so, for a single assisted result with nothing known about the borrowed part, no better metric fixes it; designs that vary tasks, assistance or people, or that bound or measure the borrowed part, are not modelled and may estimate unaided capability. Applies to a self-report only if the report is a function of assisted output, which is a layer-4 premise and is not established.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-no-procedure-on-output-recovers-fallback.md" } }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.CapabilityAssessment.protocols_are_incomparable", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.not_knowable_of_collision" ], "application": "Pricing a withdrawal-based assessment. Where one assisted result is observed and the borrowed part is unconstrained, neither protocol determines the other exactly, so within the model withdrawal testing is not redundant with observing that result; richer designs are not modelled and may recover partial or statistical information.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-protocols-are-incomparable.md" } }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.CapabilityAssessment.withdrawal_settles_and_no_output_procedure_does", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Sovereignty.fallback_not_knowable_from_assistedOutput" ], "application": "Justifying the measurement design. The expensive protocol settles the question (by definition: it measures the fallback capability itself) and no procedure on ordinary observed work does, so within the model some evidence beyond a single assisted result is needed; withdrawal testing is one such, and experimental designs that vary tasks, assistance or people are others the model does not cover.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-withdrawal-settles-and-no-output-procedure-does.md" } } ] } }, { "id": "LAND-VERIF-FULLACCESS-001", "name": "Full access to the code: Rice bounds the behavioral half and nothing else", "tags": [ "computability", "verification" ], "result_shape": "CHARACTERIZATION", "notes": "BRIDGE, AND AS MUCH A FENCE. Layer 2 is AISafetyAtlas.Verification and AISafetyAtlas.Computability, which hold Rice's theorem in two forms. This row is the arrow to Reuel, Bucknall et al. Open Problem 55 ('How can model properties be verified with full access to the model?'). THE LINE THIS ROW EXISTS TO HOLD. The natural sentence to reach for is 'no verifier exists even with full access'. THAT SENTENCE IS FALSE, and the module is built so it cannot be read off it. CONTENT. FullAccessVerifier P is a total computable procedure that reads THE CODE ITSELF and decides membership in a set of codes. The SAME verifier type appears in both theorems, so access is held fixed at the maximum and only the property varies. no_fullAccessVerifier_of_extensional: if P is EXTENSIONAL (codes with the same behaviour are alike under it) and nontrivial, no such verifier exists -- having the source does not help. That is Rice through rice_code_iff. fullAccessVerifier_exactArtifact: for the property 'this code is exactly c0', a verifier DOES exist, from full access and nothing else; it is primitive recursive and it is not meant to be deep, it is the counterexample that stops the first theorem from being read as a claim about access. ITS PROOF TOUCHES NO RICE: it goes through Mathlib's Primrec.eq and PrimrecPred.computablePred, which is the point -- in a module whose thesis is that the positive half lies outside Rice's reach, attributing that half to Rice would be the one error that matters. exactArtifact_not_extensional joins them: two distinct codes computing the same function are not alike under code identity, so Rice's hypothesis fails for it. access_is_not_what_separates_them puts both halves side by side at the same access level, which is the only honest statement: RICE BOUNDS THE BEHAVIORAL HALF OF OPEN PROBLEM 55, and what fails is not access but extensionality. WHAT IS NOT CLAIMED. NOT that verification with full access is impossible. The claim is confined to extensional properties. Properties of weights, activations, circuit structure, parameter count, training provenance or the loss landscape are NOT extensional -- two codes computing the same function can differ in all of them -- and Rice says nothing whatever about those. Those are Open Problems 21 and 22 of the same paper (mechanistic analysis of model internals, and how far it generalizes), and THIS ROW IS NOT EVIDENCE THAT THEY ARE HARD; reading it as such is the precise error it exists to prevent, and nothing here says which non-extensional properties are tractable. No real system is modelled: a Code is a Mathlib program code, and whether a trained model's weights are one in the relevant sense, and whether any property a governance regime asks about is extensional, are layer-4 assignments neither of which is made. Totality is part of the negative claim -- procedures that may abandon a case, answer only on a restricted class, or be sound in one direction only are not ruled out by anything here. RELATION TO EXISTING ROWS. LAND-VERIF-AGENTBEHAVIOR-001 carries the behavioral half at behavioral properties, via Melo et al.; this row restates it at code sets so that both halves can be compared with the verifier type and the access held fixed, and adds the positive half, which is the new content. BY-012's existing REVIEWED status covers a DIFFERENT interpretation and must not be read as covering this one. GROUNDING. Examples.Verification.FullAccess asks two questions about one source file at the same access. 'Does this program compute the constant-zero function?' is extensional by construction and nontrivial (Code.zero computes it, Code.succ does not), and no verifier answers it. 'Is this program exactly this source file?' is answered by one. Code.zero and Code.zero composed with itself are distinct codes with the same eval, which is why the second escapes Rice -- the same indistinguishability that defeats the behavioural question is what the syntactic question is allowed to see through. SOURCE STATUS. The TMLR paper is a DIRECTORY: it states open problems and no theorem, so this row claims no coverage of it and it is not in the coverage audit.", "original_source_refs": [ "reuel-bucknall-2025-open-problems-technical-ai-governance" ], "root_import": true, "public": { "group": "Limits of verification", "title": "Having the source is not the bottleneck", "summary": "Hand a checker the complete source of a program and ask two questions about it. Ask what the program *does* -- whether it computes a particular function -- and no checking procedure can always answer correctly, however much of the source it has. Ask what the program *is* -- whether it is this exact source file -- and a checker answers at once. Same source, same access, opposite outcomes. So the classical impossibility result is about questions that depend only on behaviour; it is not a statement that inspecting a system is futile, and in particular it says nothing at all about questions concerning a system's internal structure.", "use": "When a classical undecidability result is cited to argue that inspecting or auditing a system cannot work, and when deciding whether a proposed property is behavioural or structural.", "attribution": "Atlas-side bridge over Rice's theorem as carried by this repository's computability layer; the question is Open Problem 55 of Reuel, Bucknall et al., TMLR 04/2025. The result bounds only the behavioural half of that question and is not evidence about the structural half, which is the same paper's Open Problems 21 and 22" }, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Verification.FullAccess.FullAccessVerifier", "AISafetyAtlas.Verification.FullAccess.exactArtifact", "AISafetyAtlas.Verification.FullAccess.no_fullAccessVerifier_of_extensional", "AISafetyAtlas.Verification.FullAccess.fullAccessVerifier_exactArtifact", "AISafetyAtlas.Verification.FullAccess.exactArtifact_not_extensional", "AISafetyAtlas.Verification.FullAccess.access_is_not_what_separates_them" ], "module": "AISafetyAtlas.Verification.FullAccess", "atlas_module": "AISafetyAtlas.Verification.FullAccess", "build_command": "lake build AISafetyAtlas.Verification.FullAccess AISafetyAtlas.Examples.Verification.FullAccess" } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Verification.FullAccess.no_fullAccessVerifier_of_extensional", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Computability.rice_code_iff" ], "application": "Answering 'surely full access is enough'. For a nontrivial property that depends only on what a program computes, no total correct procedure decides it from the source, so obtaining the source does not by itself make such a question answerable.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-no-fullaccessverifier-of-extensional.md" } }, { "atlas_declaration": "AISafetyAtlas.Verification.FullAccess.fullAccessVerifier_exactArtifact", "type": "BRIDGE", "source_declarations": [ "Primrec.eq", "PrimrecPred.computablePred" ], "application": "Blocking the overread. A verifier for a syntactic property of the artifact exists at the same access, so an undecidability result about behaviour must not be cited as evidence that inspecting a system, or reasoning about its internal structure, cannot work.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-fullaccessverifier-exactartifact.md" } }, { "atlas_declaration": "AISafetyAtlas.Verification.FullAccess.access_is_not_what_separates_them", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Computability.rice_code_iff", "Primrec.eq" ], "application": "Classifying a proposed verification target. With the verifier type and the access held fixed, the answerable and unanswerable questions are separated by extensionality alone -- so the question to ask of a proposed property is whether it depends on behaviour only, not how much access the verifier is given.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-access-is-not-what-separates-them.md" } } ] } }, { "id": "LAND-AUDIT-LAG-001", "name": "What an audit certifies: the audited version, not the deployed one", "tags": [ "information-theory", "oversight", "verification" ], "result_shape": "POINT_IMPOSSIBILITY", "notes": "BRIDGE. Layer 2 is AISafetyAtlas.Knowledge.Temporal, which is about evidence, targets and time and names no AI system. This row is the arrow to a governance question: Reuel, Bucknall et al. Open Problem 63, 'How can it be verified that the model version on which an evaluation or audit was performed is the same as is deployed?'. CONTENT. AuditSetup is a model of an audit -- what the audit reads at each time, and which version is actually deployed at each time. audit_certifies_audited_not_deployed states both halves together, because either alone misleads: the audit determines the version it audited, AND the same evidence can fail to determine what ships. later_audit_does_not_close_the_gap does NOT refute re-auditing, and its name overstates it (its docstring says so): its second conjunct follows from the collision at the deployment time alone, and nothing in the statement ties the deployment time to the audit times. What it shows is that cumulative evidence recovering more about the PAST, by knowableFrom_mono, is compatible with a collision at the deployment time. WHAT IS NOT CLAIMED. The atlas has no model registry, no version identity and no attestation, so this is conditional on CollisionAt -- the audit's own evidence failing to separate two deployments. Read it as naming what an audit scheme must rule out, not as saying that no scheme can: a deployment publishing a signed hash of its weights has evidence this model does not, and the collision may then simply not exist. Identifying report with a real auditing methodology, or version with a real model identity, is layer 4 and is not done. GROUNDING. Examples.Knowledge.Audit inhabits both theorems on two stages and two worlds differing only in whether the operator swapped the model after the auditor left; without it the antecedent would be a hypothesis nothing satisfies. SOURCE STATUS. The TMLR paper is a DIRECTORY: it states open problems and no theorem, so this row claims no coverage of it and it is not in the coverage audit.", "original_source_refs": [ "reuel-bucknall-2025-open-problems-technical-ai-governance" ], "root_import": true, "public": { "group": "What an observer can recover", "title": "An audit tells you about the thing it audited", "summary": "An audit gathers evidence at one moment about a system that keeps changing. It settles what the system was when it was examined, and on its own need not determine what is running now -- not because auditors are careless, but because the same audit report is consistent with the audited version still running and with a different one having replaced it. Auditing again later recovers more about the past; that alone does not settle the present, and nothing here excludes later records that do.", "use": "When an audit or evaluation result is offered as evidence about a deployed system, and when deciding whether to answer that by auditing more often or by requiring the deployment to attest to what it is running.", "attribution": "Atlas-side bridge: the mathematics is this repository's temporal knowability layer; the question is Open Problem 63 of Reuel, Bucknall et al., TMLR 04/2025, which the theorem bounds and does not answer" }, "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.Audit.audit_certifies_audited_not_deployed", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.Temporal.KnowableFrom", "AISafetyAtlas.Knowledge.not_knowable_of_collision" ], "application": "Reading an audit report. A report that certifies the version it examined is evidence about that version and, on its own, no evidence about what is running now; the gap is closed only by evidence that separates the deployments, which is an attestation and not an audit finding. Stating both halves together is the point: the first (assumed, not shown) is why audits are worth performing, the second is why an audit date matters. Only the audit's own records are considered.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-audit-certifies-audited-not-deployed.md" } }, { "atlas_declaration": "AISafetyAtlas.Knowledge.Audit.later_audit_does_not_close_the_gap", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.Temporal.knowableFrom_mono", "AISafetyAtlas.Knowledge.Temporal.not_knowableAt_of_collisionAt" ], "application": "Deciding what re-auditing buys. With records that accumulate, a later audit keeps what an earlier one knew; where the records at a moment in question cannot separate two versions, an audit at that moment cannot tell which ran. Nothing here concerns a later audit about that moment: records added since may well settle it.", "review_status": "STATEMENT_REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": false, "evidence": "docs/interpretation-reviews/review-later-audit-does-not-close-the-gap.md" } } ] }, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.Audit.AuditSetup", "AISafetyAtlas.Knowledge.Audit.audit_certifies_audited_not_deployed", "AISafetyAtlas.Knowledge.Audit.later_audit_does_not_close_the_gap" ], "module": "AISafetyAtlas.Knowledge.Audit", "atlas_module": "AISafetyAtlas.Knowledge.Audit", "build_command": "lake build AISafetyAtlas.Knowledge.Audit AISafetyAtlas.Examples.Knowledge.Audit" } ] }, { "id": "LAND-INCIDENT-COUNT-001", "name": "Counting AI incidents: the number a regime publishes is a property of its filing schema", "tags": [ "information-theory", "oversight" ], "result_shape": "POINT_IMPOSSIBILITY", "notes": "BRIDGE. Layer 2 is AISafetyAtlas.Knowledge, which is about observations and decoders and names no incident. This row is the arrow to an incident-governance question: Sidhu et al. ask when a single AI incident begins and ends, and when harm traced to one model counts as one incident or many. That is an INDIVIDUATION question -- not what happened, but how many things happened -- and it looks like a definitional problem to be settled by agreeing a taxonomy. CONTENT. Reporting is a regime: what a deployment files, and how many incidents actually occurred. count_not_determined_of_collision is the obstruction in the familiar shape. count_is_not_a_measurement is the graded bridge and is about NUMBERS: under a collision, for every counting rule there are two deployments the rule scores identically whose true counts differ, so the published figure is a function of the filing schema and not of the incidents -- comparing published figures remains possible, but a figure meeting a target expressed in incidents does not show the incidents meet it, and a difference across regimes or a schema change may reflect the schemas. The quantifier over rules is the content: this is not a statement about the counting methodologies anyone has proposed. WHAT IS NOT CLAIMED. The atlas has no incident, no taxonomy and no reporting standard; nothing says any real regime collides, and exhibiting the two deployments is an empirical claim about a schema that this repository does not make. It is not an argument against counting incidents -- a count inherits the resolution of the schema it is computed from, which is a reason to specify the schema first and not a reason to stop counting. schema_fixes_the_count and knowable_of_report_carries_count are the positive half and are NOT graded BRIDGE: the first is Knowable unfolded and proves nothing beyond the definition. Identifying report with any real filing, or count with any published figure, is layer 4 and is not done. GROUNDING. Examples.Practitioner applies the bridge. SOURCE STATUS. The arXiv paper is a DIRECTORY: it poses open problems and states no theorem, so this row claims no coverage of it and it is not in the coverage audit.", "original_source_refs": [ "sidhu-etal-2026-open-problems-ai-incident-governance" ], "root_import": true, "public": { "group": "What an observer can recover", "title": "How many incidents were there?", "summary": "An incident regime publishes a count. That number is computed from what deployments file, and where two deployments file identically while differing in what actually happened, no counting rule whatever recovers the difference -- so the figure measures the filing schema rather than the incidents. It can still be compared with other figures and with targets; what the comparison does not show is a fact about incidents: a count that meets a target expressed in incidents does not show that the incidents meet it, and a difference across regimes or across a schema change may reflect the schemas rather than the incidents. Where the filing does determine the count, a rule exists -- that is the definition unfolded, and it does not say any real form determines the count; under a collision no counting rule is the repair, which is what the obstruction makes precise by quantifying over every rule.", "use": "When an incident count is published, compared across jurisdictions or time, or written into a target; and when designing what an incident report has to contain.", "attribution": "Atlas-side bridge: the mathematics is this repository's knowability kernel; the individuation question is Sidhu et al., arXiv 2026, which the theorem bounds and does not answer" }, "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.IncidentCount.count_not_determined_of_collision", "type": "WRAPPER", "source_declarations": [ "AISafetyAtlas.Knowledge.not_knowable_of_collision" ], "application": "The obstruction in the familiar shape: two deployments that file identically and differ in their incident count admit no counting rule over filings." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.IncidentCount.count_is_not_a_measurement", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.IncidentCount.count_not_determined_of_collision" ], "application": "Reading a published incident count. Under a filing collision the number is a function of the schema and not of the incidents, so a count that meets a target expressed in incidents does not show the incidents meet it, and a difference across regimes or across a schema change may reflect the schemas rather than the incidents; the quantifier is over every counting rule, so no methodology is the repair; a richer form could be only if what it records determines the true count, which presupposes what counts as one incident and complete, honest filing -- nothing here says such a form exists.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-03", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-count-is-not-a-measurement.md" } }, { "atlas_declaration": "AISafetyAtlas.Knowledge.IncidentCount.schema_fixes_the_count", "type": "WRAPPER", "source_declarations": [ "AISafetyAtlas.Knowledge.Knowable" ], "application": "The positive half: where the filing determines the count, a counting rule exists and the number means what it says. Knowable unfolded, stated so that the repair sits beside the obstruction." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.IncidentCount.knowable_of_report_carries_count", "type": "WRAPPER", "source_declarations": [ "AISafetyAtlas.Knowledge.knowable_iff_no_collision" ], "application": "A regime that files the count itself has no obstruction, which keeps the positive half from being a hypothesis nothing satisfies." } ] }, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.IncidentCount.Reporting", "AISafetyAtlas.Knowledge.IncidentCount.count_not_determined_of_collision", "AISafetyAtlas.Knowledge.IncidentCount.count_is_not_a_measurement", "AISafetyAtlas.Knowledge.IncidentCount.schema_fixes_the_count", "AISafetyAtlas.Knowledge.IncidentCount.knowable_of_report_carries_count" ], "module": "AISafetyAtlas.Knowledge.IncidentCount", "atlas_module": "AISafetyAtlas.Knowledge.IncidentCount", "build_command": "lake build AISafetyAtlas.Knowledge.IncidentCount AISafetyAtlas.Examples.Practitioner" } ] }, { "id": "LAND-KNOW-UNIFORM-001", "name": "Acting acceptably without identifying the state: the uniform decision boundary", "tags": [ "information-theory", "control-theory" ], "result_shape": "CHARACTERIZATION", "notes": "GROUNDING. Source is the unpublished governance sketch pinned in docs/provenance/governance-sketch-uniform-decision.md; not in the coverage audit, not coverage. The sketch's section 6.3, its GE5 and GE6, is the only section built from it, chosen because four of its other sections reduce to it. CONTENT. Knowable asks whether an observation determines a VALUE; this asks whether it determines an acceptable ACTION, which is strictly weaker and is what a controller actually needs. UniformlyActionable is the existential rule, FibrewiseAgreeable the per-state criterion, and they agree given a nonempty action type -- the hypothesis is print's own and is what supplies the rule's value at observations nothing realizes. uniformlyActionable_iff_iInter_nonempty restates the same boundary in print's own set form at Type. knowable_iff_uniformlyActionable is Iff.rfl, so the claim that this generalizes the kernel is a theorem and not a docstring. CORRECTION TO PRINT. GE6 presents an indistinguishable pair with disjoint acceptable sets as THE obstruction. It is sufficient and not necessary: acceptable-action sets can be pairwise intersecting and jointly empty, so the module's complete certificate is not_uniformlyActionable_iff_exists_unservable, an empty fibre, and Examples.Knowledge.UniformAction.conflict_is_not_complete proves both halves of the separation on three states. That is the one structural difference from the kernel, where equality's transitivity does make a pair complete. ROUTING. Kept here and not offered upstream: the statement is framed at an observation and an acceptability relation, which is this repository's subject, and the mathematical core is one application of choice that Mathlib has no reason to name. NEIGHBOUR, NOT INSTANCE. Sovereignty.demandwise_iff_exists_selector has the same shape but is indexed by the demand a coalition must serve rather than by what an observer sees; neither instantiates the other, and both docstrings say so. ANCHORED 2026-09-13 to Lin and Wonham 1988, read at folios 177-181. Their observability condition is pairwise and their Theorem 2.1 is still an iff, which looked at first like a contradiction of this module's incompleteness result. It is not: p.179 builds the supervisor as one binary decision per event, and pairwise agreement is complete precisely at two actions. That is now a theorem here, and the module's own three-action counterexample is its sharpness witness. A wrong generalization was refuted on the way and is recorded so it is not retried: per-coordinate binary actions do NOT make pairwise sufficient for an arbitrary predicate on a product, since {TT,TF}, {TF,FF}, {TT,FF} over Bool x Bool are pairwise intersecting and jointly empty.", "original_source_refs": [ "governance-kernel-sketch-unpublished", "lin-wonham-1988-observability" ], "related_result_ids": [ "LAND-KNOW-001", "LAND-SOV-CATALOGUE-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Knowledge.UniformlyActionable", "AISafetyAtlas.Knowledge.FibrewiseAgreeable", "AISafetyAtlas.Knowledge.ActionConflict", "AISafetyAtlas.Knowledge.uniformlyActionable_iff_fibrewiseAgreeable", "AISafetyAtlas.Knowledge.uniformlyActionable_iff_iInter_nonempty", "AISafetyAtlas.Knowledge.knowable_iff_uniformlyActionable", "AISafetyAtlas.Knowledge.not_uniformlyActionable_iff_exists_unservable", "AISafetyAtlas.Knowledge.not_uniformlyActionable_of_conflict", "AISafetyAtlas.Knowledge.UniformlyActionable.mono", "AISafetyAtlas.Knowledge.UniformlyActionable.mono_good", "AISafetyAtlas.Knowledge.uniformlyActionable_of_universal", "AISafetyAtlas.Knowledge.uniformlyActionable_of_injective", "AISafetyAtlas.Knowledge.PairwiseAgreeable", "AISafetyAtlas.Knowledge.PairwiseAgreeable.of_uniformlyActionable", "AISafetyAtlas.Knowledge.uniformlyActionable_of_pairwiseAgreeable_of_card_le_two" ], "module": "AISafetyAtlas.Knowledge.UniformAction", "atlas_module": "AISafetyAtlas.Knowledge.UniformAction", "build_command": "lake build AISafetyAtlas.Knowledge.UniformAction", "relationship": "RELATED", "scope_delta": { "summary": "Atlas-side rendering of an unpublished sketch; nothing graded against a pinned published source. Against the sketch's section 6.3: Same for the controller, the fibre-intersection criterion and both named repairs, with the sketch's own nonempty-action-type hypothesis kept. Wider in that the characterization is proved at Sort rather than Type, and in that the fibre is indexed by a state rather than by a realized observation value, which removes the realizability side condition the set form needs; wider again in carrying monotonicity in acceptability, where print names only the universally acceptable action. NARROWER in nothing. Two declarations have no counterpart in print: knowable_iff_uniformlyActionable, because print does not know this repository's kernel, and not_uniformlyActionable_iff_exists_unservable, which is the certificate that is necessary as well as sufficient. Print's GE6 is contradicted as a characterization and kept as a sufficient condition; the separating witness is in the examples module. No complexity claim, no efficiency claim, and no computation: the rule is selected using choice, which is what print says its own proof does. ANCHORING, 2026-09-13. The pairwise half now has a published anchor: Lin and Wonham 1988, whose observability condition ker P <= act_K is pairwise (p.177) and whose Theorem 2.1 (p.181) is nonetheless an iff. Graded RELATED and not EXACT, deliberately: that paper works in languages, supervisors and controllability jointly, over a discrete-event generator, where this module has one observation map, an arbitrary action type and no controllability at all. What transfers is not the theorem but the REASON -- p.179 builds the supervisor as a binary per-event decision, and uniformlyActionable_of_pairwiseAgreeable_of_card_le_two is that reason isolated: at most two actions makes pairwise agreement complete. The atlas's own three-action counterexample is then exactly the sharpness witness, so the earlier claim that the pairwise certificate is incomplete is not in tension with print but delimits it.", "evidence": "docs/provenance/governance-sketch-uniform-decision.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Knowledge.uniformlyActionable_iff_fibrewiseAgreeable", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Knowledge.UniformlyActionable", "AISafetyAtlas.Knowledge.FibrewiseAgreeable" ], "application": "Deciding whether a monitor, a reviewer or an autonomous controller has a uniform policy from what it observes. It has none exactly when some set of states it cannot separate has no action acceptable throughout; otherwise the existing observation is already enough, however ambiguous it is." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.knowable_iff_uniformlyActionable", "type": "WRAPPER", "source_declarations": [ "AISafetyAtlas.Knowledge.Knowable" ], "application": "Locating the slack between knowing and acting: the knowability kernel is this question with exactly one acceptable action per state, so every impossibility already proved about observations transfers, and each one can be asked again with a weaker demand." }, { "atlas_declaration": "AISafetyAtlas.Knowledge.not_uniformlyActionable_iff_exists_unservable", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Knowledge.uniformlyActionable_iff_fibrewiseAgreeable" ], "application": "The certificate an auditor can demand when told no safe uniform policy exists: a state whose indistinguishable neighbourhood rejects every candidate action, rather than a single awkward pair.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-04", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-not-uniformlyactionable-iff-exists-unservable.md" } }, { "atlas_declaration": "AISafetyAtlas.Knowledge.uniformlyActionable_of_pairwiseAgreeable_of_card_le_two", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Knowledge.PairwiseAgreeable", "AISafetyAtlas.Knowledge.uniformlyActionable_iff_fibrewiseAgreeable" ], "application": "Knowing when it is enough to check pairs. Auditing a monitor or controller state by state is quadratic and feasible; checking every group of confusable states at once is not. This says the cheap check suffices when the decision is binary -- act or do not act, enable or disable -- and the three-action witness shows it can fail with three actions; larger action sets are not covered by a theorem here." } ] }, "public": { "group": "What an observer can recover", "title": "You do not have to know which case you are in", "summary": "A system that cannot tell two situations apart can still act correctly in both -- it only needs one action that is acceptable in each. A fixed rule from what is observed to what is done exists exactly when every group of situations that look alike shares at least one acceptable action, and fails exactly when some such group shares none.", "use": "When a proposal is rejected because a monitor cannot identify the situation, or accepted because it can: the question is not identification but whether the situations it confuses agree on something safe to do.", "attribution": "Atlas-side interpretation of an unpublished sketch; the sketch's pairwise obstruction is kept as a sufficient condition and its converse is refuted here" }, "escape_routes": [ { "axis": "ADD_INFORMATION", "status": "FORMALIZED", "lean": "AISafetyAtlas.Knowledge.UniformlyActionable.mono", "note": "The sketch's first repair. A more informative observation keeps every uniform rule the coarser one had, because the fibres only shrink; the rule is composed with the recovery map rather than reassembled. Witnessed by Examples.Knowledge.UniformAction.uniformlyActionable_sighted, where the same two demands become servable once the observation separates the states." }, { "axis": "RELAX_EXACTNESS", "status": "FORMALIZED", "lean": "AISafetyAtlas.Knowledge.UniformlyActionable.mono_good", "note": "The sketch's second repair, and the reason this module exists. Widening what counts as acceptable keeps every rule, and uniformlyActionable_of_universal is the degenerate case a designer reaches for: one action acceptable everywhere needs no observation at all. Witnessed by Examples.Knowledge.UniformAction.fallback_repairs_a_real_failure, which pairs the repair with the failure it repairs." } ] }, { "id": "LAND-SOV-STABILITY-001", "name": "Keiding's cycle, stated at this repository's effectivity families", "tags": [ "social-choice", "multi-agent" ], "result_shape": "INFRASTRUCTURE", "notes": "GROUNDING. Keiding 1985, Definition 3.3 at p.96, read from rendered page images. WHY IT IS STATABLE HERE AT ALL. Keiding's object is E : P(N) -> P^2(A), which IS the type of AISafetyAtlas.Sovereignty.effectivity, namely Set N -> Set (Set X); GameFormAcyclic is Acyclic (effectivity G) and gameFormAcyclic_iff is Iff.rfl, so there is no translation step that could be got wrong. The condition mentions ONLY E: no preference, no utility and no ordering appears anywhere in Definition 3.3, which is what makes a purely combinatorial property decide whether a distribution of power can be blocked from every direction at once. WHAT IS HERE NOW. Theorem 3.6 -- a cycle implies E is unstable -- IS formalized as of 2026-09-16, as not_stable_of_cycle, following print's construction step for step: D_i, the relation P-bar^i, Lemma 3.7's extension (exists_wellOrder_ge_of_acyclic, assembled from Mathlib rather than reconstructed), and the player's preference read through the block an outcome lies in with the blocks outside D_i on top. acyclic_of_stable is the contrapositive under Definition 2.1(i). No separate game-form form of either is carried: gameFormAcyclic_iff is Iff.rfl, so Acyclic (effectivity G) and GameFormAcyclic G are the same proposition and a consumer applies the general theorem. Two such wrappers were written and then removed, because witnessing them needs a game form whose effectivity actually cycles -- a dictatorial form is stable -- and that is a Condorcet-style construction this row does not have. The core layer it needed -- WeakOrder, Profile, DominatesVia, Dominated, core, Stable, from Definitions 2.1 to 2.3 -- is defined here too. Theorem 3.8 is still NOT formalized: its proof runs the other way, building a cycle from an empty core, and that construction is not transcribed. TWO EARLIER COSTINGS WERE WRONG AND ARE CORRECTED. This row previously said the theorems were blocked because the repository has no preference-profile type; it has one, SocialChoice.Profile = Fin N -> Preorder' A, which is K(A)^N. The gap was the CORE, not the profiles: Sovereignty.Domination blocks a demand and mentions no preference, and TrulyPlayable.nonmonotonicCore is a different object. It also said Lemma 3.7's finiteness was unavailable; print applies 3.7 to D_i inside {1..k}, which is Fin c.len and finite for free, so not_stable_of_cycle does not assume X finite where print's A is. A FIDELITY DEFECT WAS FOUND AND FIXED. Definition 3.3 writes S_i in 2^N and p.94 defines 2^D as P(D) minus the empty set, so a cycle's coalitions are non-empty; the first transcription dropped that, which made Cycle wider than print and Theorem 3.6 false, since the empty coalition dominates nothing. Cycle.Snonempty carries it. Non-emptiness of each B_i is a hypothesis of Theorem 3.6 rather than a field, because print derives it from Definition 2.1(i), which this module does not impose on an arbitrary E. TRANSCRIPTION CHOICES. Definition 3.3(ii) quantifies over sequences of indices; List.IsChain is exactly adjacent-pairs-related, and the sequence is kept nonempty structurally as a head and a tail rather than by carrying a nonemptiness proof. Remark 3.4 says the 'or' is not exclusive and Or is not exclusive, so nothing is needed there. NOT VACUOUS ON EITHER SIDE. acyclic_bot inhabits Acyclic at the effectivity function that is effective for nothing; Examples.Sovereignty.Stability.duelCycle inhabits Cycle at two agents and two outcomes, each agent effective for exactly the outcome the other is blocked by. The chain condition there is discharged by the argument that makes cycles finite objects at all: consecutive indices must differ because a block never meets its own B, so with two agents a chain of length two visits both and two singleton coalitions intersect emptily.", "original_source_refs": [ "keiding-1985-stability-effectivity" ], "related_result_ids": [ "LAND-SOV-SERVICE-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Sovereignty.Cycle", "AISafetyAtlas.Sovereignty.Acyclic", "AISafetyAtlas.Sovereignty.acyclic_iff", "AISafetyAtlas.Sovereignty.not_acyclic_of_cycle", "AISafetyAtlas.Sovereignty.GameFormAcyclic", "AISafetyAtlas.Sovereignty.gameFormAcyclic_iff", "AISafetyAtlas.Sovereignty.not_gameFormAcyclic_of_cycle", "AISafetyAtlas.Sovereignty.acyclic_bot", "AISafetyAtlas.Sovereignty.Cycle.not_mem_block_of_mem_B", "AISafetyAtlas.Sovereignty.Cycle.ne_of_block_inter_B_nonempty", "AISafetyAtlas.Sovereignty.Cycle.Snonempty", "AISafetyAtlas.Sovereignty.Cycle.exists_block", "AISafetyAtlas.Sovereignty.Cycle.idx", "AISafetyAtlas.Sovereignty.Cycle.mem_block_idx", "AISafetyAtlas.Sovereignty.Cycle.idx_eq_of_mem", "AISafetyAtlas.Sovereignty.chainCondition_of_pairwise_disjoint", "AISafetyAtlas.Sovereignty.WeakOrder", "AISafetyAtlas.Sovereignty.WeakOrder.lt", "AISafetyAtlas.Sovereignty.WeakOrder.refl", "AISafetyAtlas.Sovereignty.Profile", "AISafetyAtlas.Sovereignty.DominatesVia", "AISafetyAtlas.Sovereignty.Dominated", "AISafetyAtlas.Sovereignty.core", "AISafetyAtlas.Sovereignty.Stable", "AISafetyAtlas.Sovereignty.mem_core_iff", "AISafetyAtlas.Sovereignty.not_mem_core_of_dominated", "AISafetyAtlas.Sovereignty.not_stable_of_core_eq_empty", "AISafetyAtlas.Sovereignty.exists_wellOrder_ge_of_acyclic", "AISafetyAtlas.Sovereignty.not_stable_of_cycle", "AISafetyAtlas.Sovereignty.NoEmptySet", "AISafetyAtlas.Sovereignty.acyclic_of_stable", "AISafetyAtlas.Sovereignty.acyclic_of_rank" ], "module": "AISafetyAtlas.Sovereignty.Stability", "atlas_module": "AISafetyAtlas.Sovereignty.Stability", "build_command": "lake build AISafetyAtlas.Sovereignty.Stability AISafetyAtlas.Examples.Sovereignty.Stability", "relationship": "EXACT", "scope_delta": { "summary": "Same as print for Definition 3.3 and for acyclicity, at the printed quantifiers, with Keiding's own generality: E is an arbitrary Set N -> Set (Set X) and neither N nor X is finite here, where print works over 2^N and 2^A without stating a finiteness hypothesis on the definition itself. WIDER in exactly that respect and in nothing else. NARROWER: Theorem 3.8 (acyclicity gives stability) is not carried. Theorem 3.6 (not_stable_of_cycle: a cycle with non-empty blocking sets makes E unstable) and Lemma 3.7 (exists_wellOrder_ge_of_acyclic) are, so the row claims the cycle-implies-unstable consequence and not its converse. The strong-cycle notion of Lemma 3.5 is not carried either, since nothing consumes it. No claim is made that acyclicity is decidable, and none that any particular game form satisfies it beyond the two witnesses.", "evidence": "docs/provenance/keiding-stability.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.not_gameFormAcyclic_of_cycle", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Sovereignty.Cycle", "AISafetyAtlas.Sovereignty.Acyclic" ], "application": "Auditing a distribution of power for instability: a cycle with nonempty blocking sets yields some preference profile under which every outcome is dominated (Keiding Theorem 3.6), so the arrangement is not stable across all profiles. A cycle is a finite certificate, checkable against the effectivity family alone, that, given nonempty blocking sets, some coalition can always block whatever is proposed -- and it never mentions what anybody wants. This is the arrow from holding such a certificate to the arrangement failing Keiding's condition." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.gameFormAcyclic_iff", "type": "BRIDGE", "source_declarations": [ "AISafetyAtlas.Sovereignty.effectivity" ], "application": "Reading the social-choice stability literature directly against a game form already in this repository: Keiding's E and the atlas's effectivity are the same type, so the condition applies with nothing in between; effectivity G meets Def. 2.1(i) only when every strategy type is inhabited.", "review_status": "STATEMENT_REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-05", "statement_reviewed": true, "interpretation_reviewed": false, "evidence": "docs/interpretation-reviews/review-gameformacyclic-iff.md" } } ] }, "public": { "group": "Aggregation and multi-agent structure", "title": "When a share-out of power cannot settle anywhere", "summary": "Some arrangements of who-can-force-what are unstable no matter what anyone wants: whatever outcome is proposed, some group can and will block it. Whether an arrangement has that defect is decided by a pattern in the powers themselves -- a cycle -- with no reference to anybody's preferences.", "use": "When checking whether a proposed governance arrangement, voting rule or veto structure can ever reach a settled outcome, before asking who prefers what.", "attribution": "Keiding (1985), Definition 3.3; the condition is transcribed, the stability theorems it feeds are not" } }, { "id": "LAND-SOV-DEONTIC-001", "name": "May, may not, can, and is empowered to are four different things", "tags": [ "ethics", "multi-agent" ], "result_shape": "INFRASTRUCTURE", "notes": "GROUNDING. Jones and Sergot 1996, abstract at p.427, read from a rendered page image: 'we distinguish institutionalised power from permission and practical possibility'. Only that separation claim is taken. THE DESIGN DECISION THIS ROW RECORDS. The repository now says 'may not', and says it as an INDEPENDENT POSITIVE PREDICATE rather than as negation of permission. NormSystem carries permitted and forbidden as two fields, which makes four statuses available instead of two: permitted, forbidden, unclassified where the norms are silent, and conflicted where they disagree. forbidden_iff_not_permitted_iff proves the four collapse to the classical two EXACTLY at consistent and complete systems, so the classical reading is a special case and nothing is lost; what is gained is that silence and contradiction become different objects and neither makes everything forbidden, which a single permitted predicate with forbidden defined as its negation cannot express. This is NOT a deontic logic: no O, no P operator, no modal axiom, no possible-worlds semantics, no inference relation. It is the minimum vocabulary in which may not can be said, and it commits the repository to nothing further. THE SEPARATION AS A THEOREM RATHER THAN A NAMING CONVENTION. Separated asks that every combination of the three axes be realized, which is the strongest reading of print's sentence; weaker readings would be satisfied by three predicates that merely differ somewhere. Examples cube_separated discharges it by decide over eight acts, one per combination. exists_empowered_possible_not_permitted is the consequence that matters for safety: an act can be institutionally effective and practically possible and still not permitted, so a system checking only can-it-be-done and does-it-count has not checked whether it is allowed. OUGHT IMPLIES CAN. Carried as a condition on a pair of predicates, not as an axiom about a modality, and deliberately NOT defined from permitted and forbidden, since that identification is a deontic-logic commitment. oughtImpliesCan_does_not_give_permission shows the principle buys no permission, using the weakest obligation that could be accused of triviality. ANCHORED 2026-09-13, second source. Grossi, Gabbay and van der Torre, chapter 7 of Dastani, Hindriks and Meyer (eds.), Springer 2010, read at folios 195, 212 and 216, supplies a published worked instance of the axis separation this row asserts. Their two implementations of one prohibition differ on exactly one axis and the difference is printed, not glossed: regimentation at folio 212 is a model update that DELETES transitions, so it acts on practical possibility, and perfect enforcement at folio 216 is DEFINED by leaving the transitions alone, its conditions opening W = W', W_end = W'_end and {R_a} = {R'_a}, so it acts on incentives and not on possibility. InstitutionalSetting.regiment and .enforce are those two moves at this row's vocabulary, with regiment_norms and enforce_possible the axis facts as rfl. The content is regiment_unpermitted_not_separated: after regimenting against exactly the unpermitted acts, possibility entails permission, so Separated FAILS -- regimentation does not record a norm, it removes the question -- while enforce_separated transports Separated unchanged. Examples.Sovereignty.Deontic.five_is_the_gap fixes act 5 of cube as the empowered, possible and unpermitted act, and shows it lost to regimentation and retained under enforcement. This repository has no payoffs, so what is taken from print is the axis claim and not the game transformation; the chapter's extensive-game framework and its retarded preconditions are recorded as not formalized. A FOURTH SOURCE, READ 2026-09-13 AND GRADED, NOT ANCHORING: Wooldridge and van der Hoek 2005, graded in section 27 of docs/provenance/source-coverage-audit.md at 0 Yes, 5 Partial, 22 No, 1 Beyond. It is the named published alternative to this module's decision to carry obligation as a PARAMETER, and reading it changed two things. FIRST, THEIR DEFINITION MAKES OBLIGATION IMPLY PERMISSION. They define P_eta phi as the grand coalition's conformant ability and O_eta phi as not-P_eta-not-phi, journal page 408; journal page 410 then gives the chain A phi -> O_eta phi -> P_eta phi -> E phi for every NON-TRIVIAL eta, with the middle link failing only at eta-top where O_eta phi holds for every phi and P_eta psi for no psi. That is vacuity, not separation. So oughtImpliesCan_does_not_give_permission is NOT an instance of their result and is not supported by it: it exists because obligation is free here, and their definition is a case where its conclusion is false. Their journal page 410 non-validities O_eta phi -> <>phi and O_eta phi -> <>phi are about GUARANTEED ability and do not refute ought-implies-can, since their own chain delivers E phi, which is the shape this row's possible has. SECOND, A CORRECTION TO THIS REPOSITORY'S OWN COSTING. docs/provenance/deontic-layer.md said adopting their definition is adopting their normative ATL. It is not: the outcome-set fragment of P_eta and O_eta needs only a restriction of conformant strategies, coalition forcing and a negation, all of which Sovereignty.Forces and Sovereignty.Enlarges already carry. ATL is needed for the TEMPORAL object, which print argues by its equation (2) is the only interesting one. The note is corrected and the fragment is costed and NOT built, because building it is answering which deontic logic the atlas adopts. ONE CONDITION OF THEIRS THIS MODULE REFUSES: journal page 402 requires every normative system to forbid whatever nature forbids, (Q minus rho(alpha)) contained in eta(alpha), coupling prohibition to possibility. cube_separated asserts the opposite - all eight combinations occur, including impossible and permitted - and that divergence is deliberate and is the price of treating silence as distinct from prohibition. THE AATS LANDED 2026-09-20 AND CLOSED SECTION 27's FIVE OWED CELLS AT ONCE. Every one of them was costed to a single missing object -- states, print's precondition function, a transition over joint actions, and infinite computations -- and AISafetyAtlas.Sovereignty.AATS is that object. Section 27 goes from 0 Yes, 5 Partial, 22 No to 10 Yes, 0 Partial, 17 No, because four rows that graded No needed only the object to be statable. TWO THINGS THE RENDERING FOUND IN PRINT. First, print's Example 2 normative system does not satisfy print's own section 3 standing requirement: nature forbids each train's move action in the crashed state and eta-one does not, so eta-one sits below eta-bottom and is not in print's lattice at all. AISafetyAtlas.Examples.Sovereignty.not_respectsNature_eta1 and not_bot_le_eta1 are those two facts; it is a slip rather than a mistake of substance, and eta1-prime is the repair. Second, print's proof of Proposition 2's converse edits a strategy to take a forbidden action and needs that edit to be legal, which is the section 3 requirement applied at the edited state; print does not cite it there and le_of_confProfile_singleton takes it as a hypothesis. THE POINT OF THE EXAMPLE IS PROVED, NOT JUST TRANSCRIBED: Examples.Sovereignty.sat_noCrash is print's stated purpose for eta-one -- that the trains never crash -- for every conformant grand-coalition profile from any uncrashed state. NOT HERE: the ⊓ and ⊔ operations and Proposition 1's lattice, sections 5 and 6, Propositions 3(2), 3(3) and 4, and any theorem relating Sovereignty.Forces to Sat at eta-bottom.", "original_source_refs": [ "grossi-gabbay-van-der-torre-2010-norm-implementation", "jones-sergot-1996-institutionalised-power", "wooldridge-van-der-hoek-2005-obligations-normative-ability" ], "related_result_ids": [ "LAND-SOV-STABILITY-001", "LAND-SOV-AUDIT-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Sovereignty.NormSystem", "AISafetyAtlas.Sovereignty.NormSystem.Unclassified", "AISafetyAtlas.Sovereignty.NormSystem.Conflicted", "AISafetyAtlas.Sovereignty.NormSystem.Consistent", "AISafetyAtlas.Sovereignty.NormSystem.Complete", "AISafetyAtlas.Sovereignty.NormSystem.consistent_iff_no_conflicted", "AISafetyAtlas.Sovereignty.NormSystem.complete_iff_no_unclassified", "AISafetyAtlas.Sovereignty.NormSystem.status_exhaustive", "AISafetyAtlas.Sovereignty.NormSystem.forbidden_iff_not_permitted_iff", "AISafetyAtlas.Sovereignty.NormSystem.not_forbidden_of_unclassified", "AISafetyAtlas.Sovereignty.InstitutionalSetting", "AISafetyAtlas.Sovereignty.InstitutionalSetting.Separated", "AISafetyAtlas.Sovereignty.InstitutionalSetting.exists_empowered_not_permitted", "AISafetyAtlas.Sovereignty.InstitutionalSetting.exists_empowered_possible_not_permitted", "AISafetyAtlas.Sovereignty.OughtImpliesCan", "AISafetyAtlas.Sovereignty.oughtImpliesCan_does_not_give_permission", "AISafetyAtlas.Sovereignty.InstitutionalSetting.regiment", "AISafetyAtlas.Sovereignty.InstitutionalSetting.enforce", "AISafetyAtlas.Sovereignty.InstitutionalSetting.regiment_norms", "AISafetyAtlas.Sovereignty.InstitutionalSetting.enforce_possible", "AISafetyAtlas.Sovereignty.InstitutionalSetting.regiment_empowered", "AISafetyAtlas.Sovereignty.InstitutionalSetting.enforce_empowered", "AISafetyAtlas.Sovereignty.InstitutionalSetting.enforce_permitted", "AISafetyAtlas.Sovereignty.InstitutionalSetting.enforce_forbidden", "AISafetyAtlas.Sovereignty.InstitutionalSetting.not_possible_of_regimented", "AISafetyAtlas.Sovereignty.InstitutionalSetting.regiment_unpermitted_not_separated", "AISafetyAtlas.Sovereignty.InstitutionalSetting.enforce_separated", "AISafetyAtlas.Sovereignty.InstitutionalSetting.regiment_and_enforce_differ" ], "module": "AISafetyAtlas.Sovereignty.Deontic", "atlas_module": "AISafetyAtlas.Sovereignty.Deontic", "build_command": "lake build AISafetyAtlas.Sovereignty.Deontic", "relationship": "RELATED", "scope_delta": { "summary": "Takes ONE sentence of print -- the abstract's separation of institutionalised power from permission and practical possibility -- and nothing else, so this is RELATED and not EXACT. NARROWER than print in almost every respect and deliberately: no conditional connective, no minimal conditional model, no action modality, no deontic logic, and none of the paper's schema rejections. WIDER in one respect: Separated demands that ALL combinations be realized, where print asserts only that the three notions are to be distinguished, and the witness discharges the stronger reading. The four-status norm annotation is NOT from this paper and is not claimed to be; it is an atlas-side design decision, justified in the row notes and collapsing to the classical two-valued reading exactly at consistent and complete systems. Obligation is left as a parameter rather than defined from permission, so nothing here takes a position on the deontic square.", "evidence": "docs/provenance/deontic-layer.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "Regime", "Detectable", "Enforces", "detectable_iff_constant_on_fibres", "not_detectable_of_indistinguishable", "unenforceable_of_indistinguishable", "undetectable_norm_is_unenforceable", "detectable_of_enforcement" ], "module": "AISafetyAtlas.Sovereignty.Enforcement", "build_command": "lake build AISafetyAtlas.Sovereignty.Enforcement", "scope_delta": { "summary": "The bridge module. AISafetyAtlas.Sovereignty.Deontic and AISafetyAtlas.Sovereignty.Auditability did not import each other; this is the edge between them and the AI-governance reading of it. Atlas-original. Witnessed in AISafetyAtlas.Examples.Sovereignty.Governance by two regimes over the same rule, one unenforceable and one enforced, so the hypothesis is doing work rather than holding vacuously.", "evidence": "docs/provenance/governance-bridges.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "ofBool", "actPairs", "findUnenforceable", "not_enforces_of_findUnenforceable_eq_some", "sanctionOf", "enforces_sanctionOf_of_findUnenforceable_eq_none" ], "module": "AISafetyAtlas.Sovereignty.EnforcementCheck", "build_command": "lake build AISafetyAtlas.Sovereignty.EnforcementCheck", "scope_delta": { "summary": "The executable half of the enforcement bridge, on the pattern Knowledge.Check and Oversight.VarietyCheck set: a function that computes and a theorem saying it agrees with the Prop. Its POSITIVE branch is constructive and is the unusual part -- a clean search returns sanctionOf, an actual monitoring rule, together with a proof that that rule sanctions every forbidden act and spares every permitted one. So the answer to an enforceable regime is the monitor rather than a verdict about monitors. Backs atlas-check kind 'enforcement'. Witnessed in AISafetyAtlas.Examples.Sovereignty.Checkers at both branches.", "evidence": "docs/provenance/practitioner-checkers.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AATS", "AATS.options", "AATS.options_nonempty", "AATS.Strategy", "AATS.nonempty_strategy", "AATS.Profile", "AATS.out", "AATS.out_univ_subsingleton", "AATS.out_univ_nonempty", "AATS.comp", "AATS.comp_univ_subsingleton", "Norm", "Norm.bot", "Norm.top", "Norm.Le", "Norm.RespectsNature", "Norm.IsNontrivial", "Norm.le_refl", "Norm.le_trans", "Norm.respectsNature_bot", "Norm.bot_le_of_respectsNature", "Norm.ofActPredicate", "AATS.Conf", "AATS.ConfProfile", "AATS.confProfile_of_le", "AATS.le_of_confProfile_singleton", "AATS.Sat", "AATS.sat_of_confProfile_mono", "AATS.sat_of_le", "AATS.sat_bot_of_respectsNature" ], "module": "AISafetyAtlas.Sovereignty.AATS", "atlas_module": "AISafetyAtlas.Sovereignty.Deontic", "build_command": "lake build AISafetyAtlas.Sovereignty.AATS", "relationship": "RELATED", "scope_delta": { "summary": "Wooldridge and van der Hoek's section 2 to section 4 at print's own definitions, built 2026-09-20. AATS is print's (n+7)-tuple with both coherence constraints as fields; the action sets' pairwise disjointness, which print states in words, is structural here because Ac is a dependent family, and print's partial tau is Option-valued with its domain fixed by print's own consistency constraint. options, Strategy with print's legality constraint, out and comp are section 2.1 and 2.2, and the grand-coalition singleton is two theorems rather than print's one remark. Norm is print's forbidden-primitive state-indexed eta, with bot, top, Le, IsNontrivial and RespectsNature. Sat is print's cooperation modality. WIDER ON FOUR AXES, none a strengthening: no finiteness of states, agents, actions or propositions; the path formula is an arbitrary set of computations, so every one of print's temporal clauses is an instance; Proposition 3(1) drops print's non-triviality, which its proof never uses; and Proposition 2's converse is stated from a weaker hypothesis than print's, at single agents rather than at every coalition. WHAT PRINT'S OWN PROOF USES AND DOES NOT CITE: Proposition 2's converse edits a strategy to take a forbidden action, and that edit is legal only by print's section 3 standing requirement, so le_of_confProfile_singleton carries RespectsNature explicitly. RELATED rather than EXACT because sections 5 and 6, Proposition 1's lattice, Propositions 3(2), 3(3) and 4 are not here, and because Sovereignty.Forces is not identified with Sat at eta-bottom.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.NormSystem.forbidden_iff_not_permitted_iff", "type": "NEW_PROOF", "source_declarations": [], "application": "Deciding whether a permissions model can be stored as one boolean per action. It can exactly when the rules are known to be both non-contradictory and exhaustive; otherwise the model silently conflates 'no rule covers this' with 'this is prohibited', which are different answers to give an operator." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.InstitutionalSetting.exists_empowered_possible_not_permitted", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Sovereignty.InstitutionalSetting.Separated" ], "application": "Reviewing an authorization check. Being able to perform an action and having the action count institutionally are two properties, and neither is permission; a gate that tests either one, or both, has still not tested whether the action is allowed." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.oughtImpliesCan_does_not_give_permission", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Sovereignty.OughtImpliesCan" ], "application": "Blocking a common argument: that because a system is required to do something and is able to do it, it is thereby allowed to. Ought-implies-can relates obligation to possibility only, so on its own it supplies no permission. The model ties obligation to no norm, so this does not block the inference in a deontic logic that also has ought-implies-may." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.Enforcement.undetectable_norm_is_unenforceable", "type": "BRIDGE", "source_declarations": [ "Enforcement.Regime", "Enforcement.Detectable", "Enforcement.Enforces", "Enforcement.unenforceable_of_indistinguishable", "Enforcement.not_detectable_of_indistinguishable", "Enforcement.detectable_of_enforcement" ], "application": "Deciding whether a rule can be made to bind. A rule closes part of the gap between what a party can do and what it may do only if some response can be conditioned on whether the rule was broken, and a response is conditioned on what a monitor sees. Where a forbidden act and a permitted act produce the same observation, no response function separates them -- the quantifier is over every response type and every function into it, so this is not a statement about the mechanisms anyone has proposed. The converse holds too: an enforcement that works is a detector, recovered from it, so enforceability and detectability are one requirement and a proposal that concedes the first while asserting the second is incoherent rather than optimistic. Given that permitted acts are never sanctioned (part of what Enforces means here), the repair named is the observation channel, never the wording of the rule or the severity of the sanction.", "review_status": "REVIEWED", "review": { "reviewer": "Mario Brcic (mbrcic)", "date": "2026-10-03", "statement_reviewed": true, "interpretation_reviewed": true, "evidence": "docs/interpretation-reviews/review-undetectable-norm-is-unenforceable.md" } } ] }, "public": { "group": "Limits of control and regulation", "title": "Allowed, able, and official are three different questions", "summary": "Whether a system may do something, whether it can, and whether the act officially counts are independent: any combination of the three can occur. Recording prohibition separately from permission also keeps 'the rules forbid this' apart from 'the rules say nothing about this', which collapse into each other if permission is stored as a single yes-or-no.", "use": "When reviewing an authorization or policy check, or when a design stores permissions as one flag per action.", "attribution": "Separation claim from Jones and Sergot (1996), abstract; the four-status norm annotation is atlas-side design" } }, { "id": "LAND-SOV-INSTITUTION-001", "name": "A Horn derivation is not counts-as, and the proof is two theorems", "tags": [ "ethics", "multi-agent", "provability-logic" ], "result_shape": "INFRASTRUCTURE", "notes": "GROUNDING. Jones and Sergot 1996, read from rendered page images at folios 427 and 431-435. WHAT THIS ROW IS FOR. Guarded Horn constitutive rules under a least fixed point are the cheapest formal object for 'the institution recognizes this act as creating this fact', and they are what an outside sketch supplied to this repository proposed as a rendering of counts-as. They are a BAD rendering of it, and this row makes that mechanical rather than advisory. Jones and Sergot deliberately reject RCM and RI at p.433 and PTR at p.435: their power is not monotone, not reflexive and does not compose. A least-fixed-point Horn derivation validates two of the three, and countsAs_refl and countsAs_trans are those two as theorems -- RI immediate from the given constructor, PTR by admissibility of cut, proved by induction on the second derivation. countsAs_validates_refl_and_trans packages both so a consumer tempted to cite that paper for CountsAs meets a theorem saying why they cannot. RCM, consequent weakening, is NOT EXPRESSIBLE here at all, since Fact is an opaque type with no entailment relation; it is neither validated nor rejected and that is recorded rather than glossed. WHAT IS NOT BUILT. The connective itself, and the minimal conditional model at p.435 whose f_s(alpha, X) : Set (Set W) is the type of effectivity. Building that is a separate decision and is not taken here. AUTHORITY. Authorized is recognized AND permitted AND practically possible. It is NOT the conjunction of LAND-SOV-DEONTIC-001's three axes: those are empowered, permitted, possible, and Authorized uses Recognized, not empowered (closure audit 2026-10-05). Recognized is also weak: Derives.given makes a fact that already holds recognized for every act. exists_recognized_not_authorized therefore asks (since 2026-10-05) for a rule concluding the fact from ground facts and firing on every empowered act, and returns an EMPOWERED, possible act recognized through that firing and not permitted -- so the separation of the axes does the work. Before, it assumed the fact already held and reduced to the existence of an act that is not permitted. no_authority_from_ungrounded_cycles is the least-fixed-point payoff: with no ground facts and no premise-free rule nothing is derivable, so rules that only cite each other install nothing.", "original_source_refs": [ "jones-sergot-1996-institutionalised-power" ], "related_result_ids": [ "LAND-SOV-DEONTIC-001" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "AISafetyAtlas.Sovereignty.ConstitutiveRule", "AISafetyAtlas.Sovereignty.Derives", "AISafetyAtlas.Sovereignty.derives_mono_base", "AISafetyAtlas.Sovereignty.no_authority_from_ungrounded_cycles", "AISafetyAtlas.Sovereignty.CountsAs", "AISafetyAtlas.Sovereignty.countsAs_refl", "AISafetyAtlas.Sovereignty.countsAs_trans", "AISafetyAtlas.Sovereignty.countsAs_validates_refl_and_trans", "AISafetyAtlas.Sovereignty.Institution", "AISafetyAtlas.Sovereignty.Institution.Recognized", "AISafetyAtlas.Sovereignty.Institution.Authorized", "AISafetyAtlas.Sovereignty.Institution.authorized_iff", "AISafetyAtlas.Sovereignty.Institution.permitted_of_authorized", "AISafetyAtlas.Sovereignty.Institution.exists_recognized_not_authorized" ], "module": "AISafetyAtlas.Sovereignty.Institution", "atlas_module": "AISafetyAtlas.Sovereignty.Institution", "build_command": "lake build AISafetyAtlas.Sovereignty.Institution", "relationship": "RELATED", "scope_delta": { "summary": "RELATED and emphatically not EXACT: this does NOT formalize Jones and Sergot's connective, and the row's main content is a proof that the object it does build is not that connective. NARROWER than print in that there is no conditional connective, no minimal conditional model, no action modality and no deontic logic. WIDER in that Event and Fact are arbitrary types with no structure imposed, where print works in a propositional modal language. DIVERGENT, and proved so rather than asserted: CountsAs validates RI and PTR, which print rejects at p.433 and p.435. RCM is not expressible, since Fact carries no entailment, so no claim is made about it in either direction. The authority decomposition is atlas-side, built on LAND-SOV-DEONTIC-001's three axes, and is not print's.", "evidence": "docs/provenance/deontic-layer.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Sovereignty.countsAs_validates_refl_and_trans", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Sovereignty.CountsAs" ], "application": "Stopping a citation that would be wrong. Rule-engine derivability is the obvious way to encode 'this act counts as that', and it silently adds reflexivity and transitivity, which the literature on institutional power rejects on purpose. A consumer reaching for that encoding now meets the two schemas as theorems." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.no_authority_from_ungrounded_cycles", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Sovereignty.Derives" ], "application": "Checking that a set of rules cannot manufacture its own authority. Rules that only cite one another derive nothing once the ground facts are removed, which is the property a delegation or ratification chain has to have." }, { "atlas_declaration": "AISafetyAtlas.Sovereignty.Institution.exists_recognized_not_authorized", "type": "NEW_PROOF", "source_declarations": [ "AISafetyAtlas.Sovereignty.InstitutionalSetting.Separated" ], "application": "Reviewing a system that treats 'the action was recognized by the rules' as sufficient. An empowered, possible act that the institution's own rule makes count as bringing about a fact can still be one nobody may perform. The theorem asks for a rule that concludes the fact from ground facts and fires on every empowered act, so recognition comes from the act, not from the fact already holding (Recognized alone would be weak: Derives.given does not look at the act)." } ] }, "public": { "group": "Limits of control and regulation", "title": "Rule engines quietly add two rules nobody asked for", "summary": "Encoding 'this action counts as that' with an ordinary rule engine adds two properties the legal-theory account deliberately excludes: everything counts as itself, and counting-as chains together. Both are proved here, so the mismatch is caught by the build rather than by a reader. Separately, an action the rules recognize is still not an action anyone is allowed to take.", "use": "When encoding institutional or organizational rules as derivations, or when a design treats rule-engine output as an authorization decision.", "attribution": "Divergence measured against Jones and Sergot (1996); the rule format and the authority decomposition are atlas-side" } }, { "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "UniformBounded", "abs_sub_le_of_monotone_discounting", "liftBddFun", "contractingWith_liftBddFun", "bddFixedPoint", "bddFixedPoint_isFixedPt", "eq_bddFixedPoint", "existsUnique_bdd_fixedPoint", "isFixedPt_mem_of_isClosed" ], "module": "AISafetyAtlas.Analysis.Blackwell", "build_command": "lake build AISafetyAtlas.Analysis.Blackwell", "scope_delta": { "summary": "The ported fixed-point core, transition-agnostic: it names no MDP, no reward and no policy. The fixed-point layer asks Nonempty and a boundedness-preservation hypothesis and nothing else, and existsUnique_bdd_fixedPoint is stated for an arbitrary operator, so the stochastic Bellman operator plugs in unchanged. The atlas adds no mathematics: the changes are the namespace, per-declaration public and @[expose] in place of a public section, the two closed-invariant-set lemmas moved out of Mathlib's ContractingWith namespace, and the three toolchain repairs listed in the row notes.", "evidence": "docs/provenance/econlib-blackwell-port.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "relationship": "RELATED", "declarations": [ "abs_expect_le", "bellmanPolicyOp_uniformBounded", "bellmanPolicyOp_mono", "bellmanPolicyOp_discounting", "bellmanPolicyOp_contractingWith_bdd", "vPiBdd", "vPiBdd_bellman", "vPiBdd_unique", "uniformBounded_of_fintype", "vPiBdd_eq_vPi" ], "module": "AISafetyAtlas.Decision.BoundedValue", "build_command": "lake build AISafetyAtlas.Decision.BoundedValue", "scope_delta": { "summary": "Atlas-original. Blackwell's two conditions discharged for the stochastic Bellman operator, and the bounded value function they yield. Nothing here restates an operator: qVal and bellmanPolicyOp lost their Fintype State binders in the same change, because neither definition ever used finiteness and only the fixed point did, so this is the same operator under a second completeness argument. Uniqueness is among bounded solutions, which is where the finiteness went, and vPiBdd_eq_vPi records that the two agree wherever both are defined.", "evidence": "docs/provenance/econlib-blackwell-port.md" } } ], "id": "LAND-BELLMAN-BDD-001", "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Decision.vPiBdd_eq_vPi", "type": "REFERENCE", "source_declarations": [ "vPiBdd", "vPiBdd_bellman", "vPiBdd_unique", "bddFixedPoint", "existsUnique_bdd_fixedPoint", "abs_sub_le_of_monotone_discounting" ], "application": "Decision theory: the discounted value of a stationary deterministic policy on a state space that need not be finite, in exchange for a uniform bound on the reward. The fixed-point core is adapted from danlyng/Econlib. This is a widening of AISafetyAtlas.Decision.DiscountedValue and not an AI-system claim: the named declaration is the theorem that the wide and narrow value functions agree wherever both are defined, and the two remaining gaps of that module -- a stationary rather than history-dependent policy, and a value defined as a fixed point rather than proved equal to a discounted return -- are untouched." } ] }, "name": "The bounded fixed point: a discounted policy value without a finite state space", "notes": "GROUNDING. AISafetyAtlas.Decision.DiscountedValue reaches vPi through ContractingWith on State -> R under the sup metric, and that is a metric space only when State is a Fintype. Gap 3 of that module's header is exactly the restriction. This row is the widening and the port it rests on. WHAT IS PROVED. AISafetyAtlas.Analysis.Blackwell carries Blackwell's sufficient condition -- monotone plus discounting gives the sup-norm contraction estimate -- and the bounded-continuous-function bridge that turns that estimate into a Banach certificate: the state type is given its own discrete topology on a type synonym, so no topology on State is required and none can collide. AISafetyAtlas.Decision.BoundedValue discharges the two conditions for the stochastic Bellman operator and defines vPiBdd, with its pointwise Bellman equation and uniqueness AMONG BOUNDED SOLUTIONS. THE JOIN. vPiBdd_eq_vPi proves the two value functions are the same function on a finite nonempty state type. Without it this would be a second development with similar names rather than a widening, which is the same standard AISafetyAtlas.Wireheading.CRMDP holds its three renderings of Theorem 11 to. WITNESSED ON AN INFINITE STATE SPACE. AISafetyAtlas.Examples.Decision.BoundedValue values a one-action walk on N at discount 1/2: constant reward gives the constant 2, and a reward paid only at state 0 gives a value that is NOT CONSTANT, so the fixed point is a function rather than a number in disguise. The narrower development cannot state either. PORT PROVENANCE. Adapted from danlyng/Econlib at 003655ccf010cdf44c4f67d6675167b54ce0e9df, Apache-2.0; per-file header and AISafetyAtlas/Upstream/LICENSE-NOTICE. THE COSTING UNDERSTATED THE DELTA AND THAT IS RECORDED RATHER THAN QUIETLY FIXED. docs/agent/policy/lean-reuse-sources.md measured the toolchain move at ONE lemma name used twice. Building it found THREE: abs_sub in the form |a - b| <= |a| + |b| does not resolve at our Mathlib, abs_add is abs_add_le, and the NNReal.coe_mk rewrite in contractingWith_liftBddFun finds no pattern because the coercion is already definitional, so a show replaces it. A fourth was a linter error: @[expose] on an abbrev has no effect. None is mathematics and all four are marked at the site; the point is that a measured cost measured by grepping names was low by a factor of three. WHAT THIS DOES NOT CLOSE. Gaps 1 and 2 of AISafetyAtlas.Decision.DiscountedValue, which no external tree bears on: vPi is a STATIONARY DETERMINISTIC policy's value rather than the carrier's history-dependent stochastic Policy, and it is DEFINED as a fixed point rather than proved equal to the discounted return along AISafetyAtlas.Decision.MDP.run. There is also no vStarBdd: the optimality operator maximises over actions with Finset.sup', so widening it is a statement about suprema over a possibly infinite action set and is a different change. NOTHING HERE IS GRADED against a printed statement and no coverage-audit section claims one. The printed reference in the upstream header, Stokey-Lucas-Prescott Corollary 1 to Theorem 3.2, is upstream's citation and is NOT PINNED here.", "original_source_refs": [ "danlyng-econlib-2026" ], "public": { "group": "Limits of control and regulation", "title": "A policy's long-run value without a finite world", "summary": "The discounted value of a fixed policy exists and is unique on a state space of any size, once rewards are bounded, and it agrees with the finite-state version wherever both apply.", "use": "When a decision model has too many states to enumerate and the question is whether its value is still well defined.", "attribution": "Blackwell's sufficient condition; Lean fixed-point core adapted from danlyng/Econlib" }, "related_result_ids": [], "root_import": true, "tags": [ "decision-theory" ] }, { "id": "LAND-GOODHART-HACKABILITY-001", "name": "Hackability: two reward functions that disagree about which policy is better", "tags": [ "agent-incentives", "decision-theory" ], "notes": "GROUNDING. Skalse, Howe, Krasheninnikov and Krueger, Defining and Characterizing Reward Hacking, NeurIPS 2022, section 4, graded in section 26 of docs/provenance/source-coverage-audit.md. CONTENT, IN TWO LAYERS. AISafetyAtlas.Decision.Occupancy is print's section 4.1: stepKernel and stateDist are the Markov chain a stationary policy induces from print's initial distribution, stateOcc is the discounted occupancy of a state, visitCount is print's F, actionEmbed is print's second embedding G, J is the pairing, and J_linear is what the whole paper runs on - value is a LINEAR FUNCTIONAL of the reward. sum_stateOcc_eq is why that is a geometric picture: total occupancy is (1 - gamma) inverse at every policy, so all policies land on one scaled simplex. AISafetyAtlas.Goodhart.Hackability is section 4.2: Hackable and Unhackable are Definition 1, Simplifies and TrivialSimplification are Definition 2, Equivalent and Trivial are print's two side conditions, and the five short theorems are print's claims about them. WIDER, AND IT IS PRINT'S OWN CONTENT: the definitions are stated on the INDUCED VALUE rather than on the reward, because print says the relation is between two reward functions mediated by the ordering each induces and neither definition mentions a reward except through its J. HackableRewards, UnhackableRewards and SimplifiesRewards are print's statements at print's arguments. A value function that comes from no reward at all is then a legitimate argument. TWO THINGS THE ATLAS DOES NOT CONNECT, BOTH COSTED IN SECTION 26. First, print defines J by the expected discounted return along a trajectory and observes it equals the pairing; this defines it by the pairing. Second, AISafetyAtlas.Decision.DiscountedValue computes the same number as a Bellman fixed point for DETERMINISTIC policies, and no theorem relates the two; docs/agent/policy/lean-parsimony.md allows a second formalization with the reason recorded, and the reason is that the bridge needs vPi lifted to stochastic policies. ONE LOOSENESS OF PRINT: footnote 5 says a trivial reward simplifies ANY reward, but Definition 2's non-degeneracy clause needs the base to be non-trivial; simplifies_of_trivial carries that hypothesis and not_simplifies_of_trivial_base shows it cannot be dropped. WITNESSES, SAME COMMIT. AISafetyAtlas.Examples.Decision.Occupancy is one state and two actions at discount one half, where stateOcc_unit gives the horizon 2 from sum_stateOcc_eq rather than from an induction. AISafetyAtlas.Examples.Goodhart.Hackability exhibits the smallest hackable pair there is - one reward pays for taking false and the other for taking true - with the constant reward as print's non-transitivity middle term. FOOTNOTE 4 IS WITNESSED ABSTRACTLY AND HAS TO BE: on one state J is affine in the action probability, so no non-trivial simplification of a non-trivial reward exists there at all, which is a small instance of what print's Theorem 3 measures and is why the library definitions are stated on the induced value. NOT HERE: Lemma 1 and Theorems 1, 2 and 3.", "original_source_refs": [ "skalse-2022-defining-characterizing-reward-hacking" ], "related_result_ids": [ "LAND-GOODHART-SELECTION-001", "BY-037" ], "root_import": true, "formalizations": [ { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "stepKernel", "stateDist", "stateDist_zero", "stateDist_succ", "stateOcc", "visitCount", "actionEmbed", "visitCount_eq_stateOcc_mul_actionEmbed", "summable_stateOcc", "stateOcc_nonneg", "stateOcc_le", "visitCount_nonneg", "sum_stateOcc_eq", "J", "J_add", "J_smul", "J_linear", "J_const", "J_const_eq" ], "module": "AISafetyAtlas.Decision.Occupancy", "build_command": "lake build AISafetyAtlas.Decision.Occupancy", "relationship": "RELATED", "scope_delta": { "summary": "Print's section 4.1 at print's objects, and WIDER on three axes print does not need: arbitrary state and action types, no reachability assumption, and no lower bound on the number of actions. Print's gamma in the unit interval is carried as gamma < 1 wherever a sum has to converge, which is print's own implicit assumption -- at gamma = 1 the visit counts of a policy that never leaves a state are infinite. RELATED rather than EXACT because print defines J by the trajectory return and observes it equals the pairing, where this defines it by the pairing; that identity has its own No row in section 26.", "evidence": "docs/provenance/source-coverage-audit.md" } }, { "framework": "Lean", "repository": "https://github.com/mbrcic/ai-safety-formalization-atlas", "version": "IN_TREE", "license": "Apache-2.0", "reproduced": true, "build_environment": "leanprover/lean4:v4.33.0; Mathlib db584cd6d46c92f209a44c0f1c829460d327499d; atlas IN_TREE", "declarations": [ "Hackable", "Unhackable", "Equivalent", "Trivial", "Simplifies", "TrivialSimplification", "Unhackable.symm", "unhackable_of_equivalent", "unhackable_of_trivial_left", "unhackable_of_trivial_right", "unhackable_not_transitive", "unhackable_of_simplifies_common", "simplifies_of_trivial", "not_simplifies_of_trivial_base", "HackableRewards", "UnhackableRewards", "SimplifiesRewards", "trivial_of_const" ], "module": "AISafetyAtlas.Goodhart.Hackability", "build_command": "lake build AISafetyAtlas.Goodhart.Hackability", "relationship": "RELATED", "scope_delta": { "summary": "Print's Definitions 1 and 2 and every claim print makes about them in section 4.2, including both footnotes. WIDER: stated on the induced value rather than on the reward, which is print's own reading of its relation; the ...Rewards family is print's statement at print's arguments. One looseness of print is corrected rather than reproduced -- footnote 5 needs the base reward to be non-trivial -- and not_simplifies_of_trivial_base shows the added hypothesis cannot be dropped. RELATED rather than EXACT because the paper's results, Lemma 1 and Theorems 1 to 3, are the geometry and are not here.", "evidence": "docs/provenance/source-coverage-audit.md" } } ], "lean_artifact": { "declarations": [ { "atlas_declaration": "AISafetyAtlas.Decision.J_linear", "type": "NEW_PROOF", "source_declarations": [ "visitCount", "actionEmbed", "J", "J_linear", "sum_stateOcc_eq" ], "application": "Reward specification: a policy is a point in occupancy space and a reward function is a linear functional on it, so two rewards agree about which policy is better exactly when they agree on a linear comparison. Skalse et al. 2022, section 4.1." }, { "atlas_declaration": "AISafetyAtlas.Goodhart.unhackable_of_simplifies_common", "type": "NEW_PROOF", "source_declarations": [ "Hackable", "Unhackable", "Simplifies", "unhackable_not_transitive", "unhackable_of_simplifies_common", "simplifies_of_trivial" ], "application": "Reward specification: whether a proxy can disagree with the true objective about which of two policies is better, and what survives composing such comparisons. Skalse et al. 2022, Definitions 1 and 2 and footnotes 4 and 5." } ] }, "public": { "group": "What an observer can recover", "title": "When a proxy reward can disagree", "summary": "Two reward functions are hackable when some change of policy looks like an improvement to one and a loss to the other.", "use": "When checking whether a proxy objective can be optimized against the objective it stands in for.", "attribution": "Skalse, Howe, Krasheninnikov & Krueger" } } ] }