From 62ed473642b528cad6ea2bb47ae34227db336e51 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 05:13:01 -0400 Subject: [PATCH 01/38] Make the open-statement policy configurable (spec A1-A5) Add an `open_statements: allowed|forbidden` policy to roadmap/README.md. Absent or forbidden keeps the strict policy, where CI rejects every sorry; the key on any other article, or any other value, is a validation issue. Derived status now records can_state, can_prove, waiting_on and assumes. Under the strict policy a statement also waits for its proof prerequisites to be proved, matching what `work list` already enforced. Under the open policy a statement waits only for its statement prerequisites, a proof for every prerequisite to be stated, and a proof that rests on an open statement gets the new violet `conditional` state ("conditionally proved"), never fully proved. The legend, the Next up card and article pages follow: conditional articles carry an Assumes row naming the open statements they rest on. The runtime projection takes readiness from the derived status and serializes assumes, waiting_on and the policy. The work frontier reads blockers from the derived status and names the policy and each item's assumptions in text and JSON. The new `autoform work assumptions [target] [--json]` command emits the autoform-assumptions/v1 contract that CI audits the Lean build against. Strict-policy work text is unchanged. --- autoform_cli/__main__.py | 43 ++++- autoform_cli/graph.py | 21 ++- autoform_cli/mermaid.py | 5 +- autoform_cli/render.py | 41 +++-- autoform_cli/runtime.py | 21 ++- autoform_cli/status.py | 72 +++++++- autoform_cli/work.py | 120 ++++++++++--- tests/test_graph.py | 51 ++++++ tests/test_render.py | 55 ++++++ tests/test_runtime.py | 58 +++++++ tests/test_status.py | 164 +++++++++++++++++- tests/test_visualization.py | 21 +++ tests/test_work.py | 337 +++++++++++++++++++++++++++++++++++- 13 files changed, 955 insertions(+), 54 deletions(-) diff --git a/autoform_cli/__main__.py b/autoform_cli/__main__.py index 45b4510e..722ba1aa 100644 --- a/autoform_cli/__main__.py +++ b/autoform_cli/__main__.py @@ -34,7 +34,7 @@ write_packets, write_skeleton_report, ) -from .work import WORK_SCHEMA, WorkError, list_ready_work, work_context +from .work import WORK_SCHEMA, WorkError, assumption_contract, list_ready_work, work_context def main(argv: Sequence[str] | None = None) -> int: @@ -132,6 +132,13 @@ def main(argv: Sequence[str] | None = None) -> int: work_context_parser.add_argument( "--json", action="store_true", help="write stable machine-readable output" ) + work_assumptions = work_subparsers.add_parser( + "assumptions", help="list open statements and the open statements each stated article rests on" + ) + work_assumptions.add_argument( + "target", nargs="?", default=".", help="project root or blueprint directory" + ) + work_assumptions.add_argument("--json", action="store_true", help="write stable machine-readable output") claim = subparsers.add_parser("claim", help="coordinate temporary node ownership through Git refs") claim_subparsers = claim.add_subparsers(dest="claim_command", required=True) for operation in ("acquire", "renew", "release"): @@ -429,6 +436,8 @@ def _project(args: argparse.Namespace) -> int: def _work(args: argparse.Namespace) -> int: + if args.work_command == "assumptions": + return _work_assumptions(args) # Only loading the roadmap can fail on the project's paths; printing the # result stays outside, so an output error is not reported as one. try: @@ -455,12 +464,16 @@ def _work(args: argparse.Namespace) -> int: if args.json: print(frontier.to_json()) return 0 + if frontier.open_statements: + print("Open statements: allowed (a statement may land with a sorry proof)") if not frontier.items: print("No ready formalization work.") return 0 for item in frontier.items: durable = f" [{item.article_id}]" if item.article_id else "" print(_human_text(f"{item.phase}: {item.node_id}{durable} - {item.title}")) + if item.assumes: + print(_human_text(" assumes: " + ", ".join(item.assumes))) return 0 if args.json: @@ -485,6 +498,10 @@ def _work(args: argparse.Namespace) -> int: print(_human_text(f"{item.title} ({item.node_id})")) print(f"State: {item.state}") print(f"Phase: {phase}") + if item.open_statements: + print("Open statements: allowed") + if item.assumes: + print(_human_text("Assumes: " + ", ".join(item.assumes))) print(_human_text(f"Claim target: {item.claim_target}")) if item.blockers: print(_human_text("Blocked by: " + ", ".join(item.blockers))) @@ -501,6 +518,30 @@ def _work(args: argparse.Namespace) -> int: return 0 +def _work_assumptions(args: argparse.Namespace) -> int: + # As in `_work`, only loading the roadmap is reported as a path error. + try: + contract = assumption_contract(args.target) + except (GraphValidationError, RuntimeProjectionError) as error: + for issue in error.issues: + print(f"error: {_human_text(issue)}", file=sys.stderr) + return 2 + except (OSError, RuntimeError, ValueError): + print("error: project or blueprint path cannot be read", file=sys.stderr) + return 2 + + if args.json: + print(contract.to_json()) + return 0 + print(_human_text(f"Open statements: {'allowed' if contract.open_statements else 'forbidden'}")) + for article in contract.articles: + if article.open: + print(_human_text(f"open: {article.id} ({', '.join(article.declarations)})")) + if article.assumes: + print(_human_text(f"conditional: {article.id} assumes {', '.join(article.assumes)}")) + return 0 + + def _print_project_inspection(result) -> None: if result.project_root is not None: print(f"Project root: {_human_text(result.project_root)}") diff --git a/autoform_cli/graph.py b/autoform_cli/graph.py index 1982dd73..916507a4 100644 --- a/autoform_cli/graph.py +++ b/autoform_cli/graph.py @@ -35,6 +35,7 @@ "not_ready", "origin", "discussion", + "open_statements", } ) _FORMALIZED = "formalized" @@ -102,6 +103,10 @@ class Graph: blueprint_dir: Path nodes: dict[str, Node] + #: ``open_statements: allowed`` in ``roadmap/README.md``: a theorem's + #: statement may land with a ``sorry`` proof. Absent or ``forbidden`` keeps + #: the strict policy, where CI rejects every ``sorry``. + open_statements: bool = False @property def edge_count(self) -> int: @@ -146,6 +151,8 @@ def load_graph(blueprint_dir: str | Path) -> Graph: issues.extend(discovery_issues) article_ids: dict[str, str] = {} source_hashes = {source.id: source.source_sha256 for source in sources} + policy_page = (blueprint / "roadmap" / "README.md").resolve() + open_statements = False for source in sources: canonical = source.path.resolve() @@ -173,6 +180,14 @@ def load_graph(blueprint_dir: str | Path) -> Graph: ) else: article_ids[article_id] = node.id + policy = node.metadata.get("open_statements") + if policy is not None: + if canonical != policy_page: + issues.append( + f"{node.id}: open_statements is a project policy; set it only in roadmap/README.md" + ) + else: + open_statements = policy == "allowed" parsed.append(node) if issues: @@ -232,7 +247,7 @@ def resolve(targets: tuple[str, ...], node: _ParsedNode = parsed_node) -> list[s issues.extend(_find_rollup_cycles(nodes)) if issues: raise GraphValidationError(issues) - return Graph(blueprint_dir=blueprint, nodes=nodes) + return Graph(blueprint_dir=blueprint, nodes=nodes, open_statements=open_statements) def _discover_nodes(blueprint: Path) -> tuple[list[_NodeSource], list[str]]: @@ -480,6 +495,10 @@ def _normalize_value(node_id: str, line_number: int, key: str, value: str) -> tu if folded not in {"cited", "bridged", "background"}: return value, f"{location}: 'origin' accepts cited, bridged, or background" return folded, None + if key == "open_statements": + if folded not in {"allowed", "forbidden"}: + return value, f"{location}: 'open_statements' accepts allowed or forbidden" + return folded, None return value, None diff --git a/autoform_cli/mermaid.py b/autoform_cli/mermaid.py index a9f92ef2..93d051a4 100644 --- a/autoform_cli/mermaid.py +++ b/autoform_cli/mermaid.py @@ -265,9 +265,10 @@ def render_legend_tip(statuses: dict[str, NodeStatus]) -> str: "fully_proved": "Proved, and every prerequisite is fully proved too.", "proved": "Proof compiles, but something it rests on is not finished.", "defined": "Definition is written in Lean.", - "can_prove": "Statement is in Lean and every prerequisite is proved — ready to work.", + "conditional": "Proof compiles, but rests on an open statement whose Lean proof is still sorry.", + "can_prove": "Statement is in Lean and nothing it needs is blocked, so the proof can start.", "stated": "Statement is in Lean; the proof is not.", - "can_state": "Prerequisites are stated, so this can be written down.", + "can_state": "Nothing it needs is blocked, so the statement can be written in Lean.", "not_ready": "Needs more blueprint work before it can be attempted.", "planned": "Described in the blueprint only.", } diff --git a/autoform_cli/render.py b/autoform_cli/render.py index bd514428..7891b9a4 100644 --- a/autoform_cli/render.py +++ b/autoform_cli/render.py @@ -856,9 +856,9 @@ def _next_target( f'{title}' if statement else title ) why = ( - "Every prerequisite is proved, so the proof can be written now." + "Its prerequisites are ready, so the proof can be written now." if node_status.key == "can_prove" - else "Every prerequisite is stated, so this can be written down." + else "Its prerequisites are ready, so the statement can be written down." ) actions = [f'Dependencies'] if chapter_page is not None: @@ -1702,6 +1702,13 @@ def _render_environment( context_link = _graph_context_link(node, page=page, destination=destination) source_link = _vault_source_link(node, repo_root=repo_root, linker=linker) meta_rows = implementation_rows + if node_status.key == "conditional": + # A conditional proof must never read as finished, so the open + # statements it rests on are named on the statement itself. + assumed = _node_references( + node_status.assumes, graph=graph, statuses=statuses, numbers=numbers, links=links + ) + meta_rows.append(("Assumes", f"{assumed} (open statements whose Lean proofs are still sorry)")) if node.discussion: meta_rows.append(("Discussion", _discussion_link(node.discussion, linker))) meta = _render_rows(meta_rows, css_class="bp-meta") @@ -1849,15 +1856,7 @@ def _dependency_disclosure( rows: list[tuple[str, str]] = [] def references(node_ids: list[str] | tuple[str, ...]) -> str: - rendered = [] - for other_id in node_ids: - other = graph.nodes[other_id] - label = html.escape(f"{numbers[other_id]} ({other.title})") - rendered.append( - f'{label}' - ) - return " · ".join(rendered) + return _node_references(node_ids, graph=graph, statuses=statuses, numbers=numbers, links=links) if node.statement_dependencies: rows.append(("Statement uses", references(node.statement_dependencies))) @@ -1873,6 +1872,26 @@ def references(node_ids: list[str] | tuple[str, ...]) -> str: return f'
Dependencies{body}
' +def _node_references( + node_ids: list[str] | tuple[str, ...], + *, + graph: Graph, + statuses: dict[str, status.NodeStatus], + numbers: dict[str, str], + links: dict[str, str], +) -> str: + """Link each node by number and title, coloured by its derived state.""" + rendered = [] + for other_id in node_ids: + other = graph.nodes[other_id] + label = html.escape(f"{numbers[other_id]} ({other.title})") + rendered.append( + f'{label}' + ) + return " · ".join(rendered) + + def _render_rows(rows: list[tuple[str, str]], *, css_class: str) -> str: if not rows: return "" diff --git a/autoform_cli/runtime.py b/autoform_cli/runtime.py index 16559104..49db6b8f 100644 --- a/autoform_cli/runtime.py +++ b/autoform_cli/runtime.py @@ -64,9 +64,12 @@ class RuntimeStatus: proved: bool fully_proved: bool defined: bool + assumes: tuple[str, ...] + waiting_on: tuple[str, ...] - def as_dict(self) -> dict[str, bool | str]: + def as_dict(self) -> dict[str, bool | str | list[str]]: return { + "assumes": list(self.assumes), "can_prove": self.can_prove, "can_state": self.can_state, "defined": self.defined, @@ -74,6 +77,7 @@ def as_dict(self) -> dict[str, bool | str]: "proved": self.proved, "state": self.state, "stated": self.stated, + "waiting_on": list(self.waiting_on), } @@ -157,6 +161,7 @@ class RuntimeGraph: dispatchable_count: int dependency_count: int maximum_depth: int + open_statements: bool = False def get(self, node_id: str) -> RuntimeNode | None: """Return a node without exposing mutable lookup state.""" @@ -175,6 +180,7 @@ def as_dict(self) -> dict[str, object]: "formalizable_count": self.formalizable_count, "maximum_depth": self.maximum_depth, "nodes": [node.as_dict() for node in self.nodes], + "open_statements": self.open_statements, "schema": self.schema, "source_revision": self.source_revision, } @@ -317,12 +323,6 @@ def build_runtime_graph( for node_id in sorted(graph.nodes): node = graph.nodes[node_id] node_status = statuses[node_id] - can_state = all(statuses[dependency].stated for dependency in node.statement_dependencies) - can_prove = ( - node_status.stated - and can_state - and all(statuses[dependency].proved for dependency in node.proof_dependencies) - ) lean_targets: list[RuntimeLeanTarget] = [] for name in declaration_names(node.lean or ""): declaration = lean_index.find(name) if lean_index is not None else None @@ -352,12 +352,14 @@ def build_runtime_graph( ), status=RuntimeStatus( state=node_status.key, - can_state=can_state, - can_prove=can_prove, + can_state=node_status.can_state, + can_prove=node_status.can_prove, stated=node_status.stated, proved=node_status.proved, fully_proved=node_status.fully_proved, defined=is_definition(node) and node_status.stated, + assumes=node_status.assumes, + waiting_on=node_status.waiting_on, ), origin=node.origin, source_targets=node.sources, @@ -380,6 +382,7 @@ def build_runtime_graph( dispatchable_count=sum(node.dispatchable for node in nodes), dependency_count=sum(len(node.dependencies) for node in nodes), maximum_depth=max((node.depth for node in nodes), default=0), + open_statements=graph.open_statements, ) _validate_runtime(runtime) return runtime diff --git a/autoform_cli/status.py b/autoform_cli/status.py index 821950e4..032076d5 100644 --- a/autoform_cli/status.py +++ b/autoform_cli/status.py @@ -65,7 +65,10 @@ class State: #: semantic set -- #31A24C green, #0064E0 blue, #F7B928 amber, #B0B3B8 grey -- #: so that finished, actionable, blocked and untouched read at a glance without #: anyone learning a legend. Green tracks proof progress, blue marks what a -#: contributor can pick up now, amber marks what nothing can start on. +#: contributor can pick up now, amber marks what nothing can start on. Violet +#: marks a proof that compiles but rests on an open statement (a theorem whose +#: Lean proof is still ``sorry``): filled because the work is done, never green +#: so that it cannot be mistaken for a sorry-free proof. #: #: Dark is not light dimmed: on #18191A a saturated fill closes up, so dark #: states are near-black panels with a bright stroke and brighter label. @@ -78,6 +81,8 @@ class State: "#122A1B", "#31A24C", "#6BD97F"), State("defined", "defined", "#C3E9CE", "#22773A", "#0B2415", "#122A1B", "#2B8F44", "#6BD97F"), + State("conditional", "conditionally proved", "#E9DFFC", "#6B3FCF", "#2E1065", + "#231A36", "#9F7AEA", "#D6C8FA"), State("can_prove", "ready to prove", "#FFFFFF", "#0064E0", "#0064E0", "#101F33", "#2D88FF", "#7FB8FF"), State("stated", "statement formalized", "#FFFFFF", "#31A24C", "#22773A", @@ -95,13 +100,23 @@ class State: @dataclass(frozen=True, slots=True) class NodeStatus: - """The derived progress of a single node.""" + """The derived progress of a single node. + + ``waiting_on`` names the prerequisites that keep an unproved node from its + next phase, in authored order. ``assumes`` names the open statements (stated, + not proved) a proof of this node rests on; it is empty unless the project + allows open statements. + """ node_id: str state: State stated: bool proved: bool fully_proved: bool + can_state: bool = False + can_prove: bool = False + assumes: tuple[str, ...] = () + waiting_on: tuple[str, ...] = () @property def key(self) -> str: @@ -121,8 +136,22 @@ def is_definition(node: Node) -> bool: def derive(graph: Graph) -> dict[str, NodeStatus]: - """Return the derived status of every node in *graph*, keyed by node id.""" + """Return the derived status of every node in *graph*, keyed by node id. + + Readiness follows the project's policy. Under the default strict policy CI + rejects every ``sorry``, so a theorem's statement lands only with its proof: + both phases wait for the statement prerequisites to be stated and the proof + prerequisites to be proved. When ``roadmap/README.md`` sets + ``open_statements: allowed``, a statement may land with a ``sorry`` proof, so + a statement waits only for its statement prerequisites and a proof for every + prerequisite to be stated. + """ + open_policy = graph.open_statements statuses: dict[str, NodeStatus] = {} + # The open statements a proof reaches through each node, as the Lean walk + # would: an open statement contributes itself and whatever its statement + # reaches, and a proved node is entered fully. + reaches: dict[str, frozenset[str]] = {} for node_id in topological_order(graph): node = graph.nodes[node_id] definition = is_definition(node) @@ -134,8 +163,33 @@ def done(dependency_id: str, attribute: str) -> bool: dependency = statuses.get(dependency_id) return dependency is not None and getattr(dependency, attribute) - can_state = all(done(other, "stated") for other in node.statement_dependencies) - can_prove = can_state and all(done(other, "proved") for other in node.proof_dependencies) + unstated = [other for other in node.statement_dependencies if not done(other, "stated")] + # A sorry proof compiles under the open policy, so there a proof + # prerequisite only has to be stated. + needed = "stated" if open_policy else "proved" + unmet: list[str] = [] + for other in node.proof_dependencies: + if other not in unstated and other not in unmet and not done(other, needed): + unmet.append(other) + assumes: tuple[str, ...] = () + if open_policy: + can_state = not unstated + can_prove = stated and not unstated and not unmet + waiting = unstated + unmet if stated else unstated + # Mathlib and unstated nodes reach nothing, so they get no entry. + if not node.mathlib: + reached = frozenset().union(*(reaches.get(other, frozenset()) for other in node.dependencies)) + assumes = tuple(sorted(reached)) + if proved: + reaches[node_id] = reached + elif stated: + reaches[node_id] = frozenset({node_id}).union( + *(reaches.get(other, frozenset()) for other in node.statement_dependencies) + ) + else: + can_state = not unstated and not unmet + can_prove = stated and can_state + waiting = unstated + unmet fully_proved = proved and all( done(other, "fully_proved") for other in node.dependencies ) @@ -150,11 +204,16 @@ def done(dependency_id: str, attribute: str) -> bool: fully_proved=fully_proved, can_state=can_state, can_prove=can_prove, + assumes=assumes, ) ], stated=stated, proved=proved, fully_proved=fully_proved, + can_state=can_state, + can_prove=can_prove, + assumes=assumes, + waiting_on=() if proved else tuple(waiting), ) return statuses @@ -168,12 +227,15 @@ def _classify( fully_proved: bool, can_state: bool, can_prove: bool, + assumes: tuple[str, ...], ) -> str: if node.mathlib: return "mathlib" if fully_proved: return "fully_proved" if proved: + if assumes: + return "conditional" return "defined" if definition else "proved" if stated: return "can_prove" if can_prove else "stated" diff --git a/autoform_cli/work.py b/autoform_cli/work.py index d7c27fed..16a281d6 100644 --- a/autoform_cli/work.py +++ b/autoform_cli/work.py @@ -10,6 +10,7 @@ WORK_SCHEMA = "autoform-work/v1" +ASSUMPTIONS_SCHEMA = "autoform-assumptions/v1" class WorkError(ValueError): @@ -42,6 +43,8 @@ class WorkItem: dependencies: tuple[str, ...] source_targets: tuple[str, ...] lean_targets: tuple[WorkLeanTarget, ...] + assumes: tuple[str, ...] = () + open_statements: bool = False @property def ready(self) -> bool: @@ -52,11 +55,13 @@ def as_dict(self) -> dict[str, object]: "article_id": self.article_id, "article_path": self.article_path, "article_revision": self.article_revision, + "assumes": list(self.assumes), "blockers": list(self.blockers), "claim_target": self.claim_target, "dependencies": list(self.dependencies), "lean_targets": [target.as_dict() for target in self.lean_targets], "node_id": self.node_id, + "open_statements": self.open_statements, "phase": self.phase, "ready": self.ready, "source_targets": list(self.source_targets), @@ -69,10 +74,12 @@ def as_dict(self) -> dict[str, object]: class WorkFrontier: source_revision: str items: tuple[WorkItem, ...] + open_statements: bool = False def as_dict(self) -> dict[str, object]: return { "schema": WORK_SCHEMA, + "open_statements": self.open_statements, "source_revision": self.source_revision, "items": [item.as_dict() for item in self.items], } @@ -87,7 +94,7 @@ def _phase(node: RuntimeNode, blockers: tuple[str, ...]) -> str | None: return "proof" if node.status.stated else "statement" -def _blockers(nodes: dict[str, RuntimeNode], node: RuntimeNode) -> tuple[str, ...]: +def _blockers(node: RuntimeNode) -> tuple[str, ...]: # Report each reason where `list_ready_work` enforces it: finished articles # need no metadata, and an unfinished leaf needs it even when not ready. if not node.dispatchable: @@ -103,24 +110,14 @@ def _blockers(nodes: dict[str, RuntimeNode], node: RuntimeNode) -> tuple[str, .. return tuple(metadata_blockers) if node.assertions.not_ready: return ("roadmap:not-ready",) - # Project CI rejects `sorry`, so a theorem's statement can only land with its - # proof: both phases wait for the proof prerequisites as well. - blocked = [ - dependency - for dependency in node.statement_dependencies - if (resolved := nodes.get(dependency)) is None or not resolved.status.stated - ] - blocked.extend( - dependency - for dependency in node.proof_dependencies - if dependency not in blocked - and ((resolved := nodes.get(dependency)) is None or not resolved.status.proved) - ) - return tuple(blocked) + # Readiness comes from the derived status, which applies the project's + # open-statement policy: strict projects wait for proof prerequisites to be + # proved, open-statement projects only for prerequisites to be stated. + return node.status.waiting_on -def _item(nodes: dict[str, RuntimeNode], node: RuntimeNode) -> WorkItem: - blockers = _blockers(nodes, node) +def _item(node: RuntimeNode, *, open_statements: bool) -> WorkItem: + blockers = _blockers(node) return WorkItem( node_id=node.id, article_id=node.article_id, @@ -137,6 +134,8 @@ def _item(nodes: dict[str, RuntimeNode], node: RuntimeNode) -> WorkItem: WorkLeanTarget(target.declaration, target.source_file) for target in node.lean_targets ), + assumes=node.status.assumes, + open_statements=open_statements, ) @@ -174,13 +173,12 @@ def list_ready_work( "formalizable leaves need durable article revision metadata: " + ", ".join(unversioned) ) - nodes = {node.id: node for node in runtime.nodes} items = tuple( item for node in runtime.nodes - if (item := _item(nodes, node)).ready + if (item := _item(node, open_statements=runtime.open_statements)).ready ) - return WorkFrontier(runtime.source_revision, items) + return WorkFrontier(runtime.source_revision, items, open_statements=runtime.open_statements) def work_context( @@ -190,7 +188,6 @@ def work_context( lean_root: str | Path | None = None, ) -> tuple[str, WorkItem]: runtime = load_runtime_graph(project_or_blueprint, lean_root=lean_root) - nodes = {node.id: node for node in runtime.nodes} matches = [ node for node in runtime.nodes @@ -206,15 +203,94 @@ def work_context( f"{selector!r} matches more than one article: " + ", ".join(node.id for node in matches) ) - return runtime.source_revision, _item(nodes, matches[0]) + return runtime.source_revision, _item(matches[0], open_statements=runtime.open_statements) + + +@dataclass(frozen=True, slots=True) +class AssumptionArticle: + id: str + article_id: str | None + state: str + declarations: tuple[str, ...] + open: bool + assumes: tuple[str, ...] + allowed_open_declarations: tuple[str, ...] + + def as_dict(self) -> dict[str, object]: + return { + "allowed_open_declarations": list(self.allowed_open_declarations), + "article_id": self.article_id, + "assumes": list(self.assumes), + "declarations": list(self.declarations), + "id": self.id, + "open": self.open, + "state": self.state, + } + + +@dataclass(frozen=True, slots=True) +class AssumptionContract: + open_statements: bool + source_revision: str + articles: tuple[AssumptionArticle, ...] + + def as_dict(self) -> dict[str, object]: + return { + "schema": ASSUMPTIONS_SCHEMA, + "open_statements": self.open_statements, + "source_revision": self.source_revision, + "articles": [article.as_dict() for article in self.articles], + } + + def to_json(self) -> str: + return json.dumps(self.as_dict(), sort_keys=True, separators=(",", ":")) + + +def assumption_contract(project_or_blueprint: str | Path) -> AssumptionContract: + """Record which open statements each stated article's Lean code may reach. + + CI audits the Lean build against this: an open article's own declarations + may keep a ``sorry`` proof, and every other article may reach only the open + statements its Markdown dependencies declare. Under the strict policy no + article is open and nothing is allowed. + """ + runtime = load_runtime_graph(project_or_blueprint) + declarations = { + node.id: tuple(target.declaration for target in node.lean_targets) + for node in runtime.nodes + } + articles: list[AssumptionArticle] = [] + for node in sorted(runtime.nodes, key=lambda candidate: candidate.id): + if node.mathlib or not node.status.stated or not declarations[node.id]: + continue + is_open = runtime.open_statements and node.status.stated and not node.status.proved + allowed = {name for assumed in node.status.assumes for name in declarations.get(assumed, ())} + if is_open: + allowed.update(declarations[node.id]) + articles.append( + AssumptionArticle( + id=node.id, + article_id=node.article_id, + state=node.status.state, + declarations=declarations[node.id], + open=is_open, + assumes=node.status.assumes, + allowed_open_declarations=tuple(sorted(allowed)), + ) + ) + return AssumptionContract(runtime.open_statements, runtime.source_revision, tuple(articles)) __all__ = [ + "ASSUMPTIONS_SCHEMA", "WORK_SCHEMA", + "AssumptionArticle", + "AssumptionContract", "WorkError", "WorkFrontier", "WorkItem", "WorkLeanTarget", + "assumption_contract", "list_ready_work", "work_context", ] diff --git a/tests/test_graph.py b/tests/test_graph.py index c265fa6b..7d838a41 100644 --- a/tests/test_graph.py +++ b/tests/test_graph.py @@ -399,3 +399,54 @@ def test_a_chapter_whose_articles_are_all_in_buckets_is_still_refused(tmp_path: load_graph(tmp_path / "blueprint") assert "orphan: chapter directory holds 1 article(s) but no README.md" in str(caught.value) + + +@pytest.mark.parametrize( + ("value", "allowed"), + [ + (None, False), + ("allowed", True), + ("forbidden", False), + ('"allowed"', True), + ("'forbidden'", False), + ("ALLOWED", True), + ("Forbidden", False), + ], +) +def test_open_statements_is_read_from_the_roadmap_root( + tmp_path: Path, value: str | None, allowed: bool +) -> None: + """Absent means forbidden; values unquote and casefold like every other scalar.""" + blueprint = tmp_path / "blueprint" + policy = "" if value is None else f"open_statements: {value}\n" + _roadmap_page(blueprint, "README.md", f"---\n{policy}---\n\n# Roadmap\n") + _node(blueprint, "result.md", "# Result\n", declaration="theorem") + + assert load_graph(blueprint).open_statements is allowed + + +def test_rejects_an_open_statements_value_other_than_allowed_or_forbidden(tmp_path: Path) -> None: + blueprint = tmp_path / "blueprint" + _roadmap_page(blueprint, "README.md", "---\nopen_statements: yes\n---\n\n# Roadmap\n") + + with pytest.raises(GraphValidationError) as caught: + load_graph(blueprint) + + assert caught.value.issues == ("roadmap:2: 'open_statements' accepts allowed or forbidden",) + + +@pytest.mark.parametrize(("relative", "node_id"), [("result.md", "result"), ("chapter/README.md", "chapter")]) +def test_open_statements_set_outside_the_roadmap_root_is_refused( + tmp_path: Path, relative: str, node_id: str +) -> None: + """The policy belongs to the project, so a chapter or article cannot opt in on its own.""" + blueprint = tmp_path / "blueprint" + _roadmap_page(blueprint, "README.md", "---\n---\n\n# Roadmap\n") + _node(blueprint, relative, "# Result\n", open_statements="forbidden") + + with pytest.raises(GraphValidationError) as caught: + load_graph(blueprint) + + assert caught.value.issues == ( + f"{node_id}: open_statements is a project policy; set it only in roadmap/README.md", + ) diff --git a/tests/test_render.py b/tests/test_render.py index 0eb5d7a0..ae5cfea8 100644 --- a/tests/test_render.py +++ b/tests/test_render.py @@ -1275,3 +1275,58 @@ def test_a_node_that_is_the_current_page_links_as_a_bare_fragment(tmp_path: Path "chapter": "#", "chapter/x": "#x", } + + +def _conditional_project(tmp_path: Path, policy: str) -> Path: + """`_project` with Top proved from an open statement, under the given policy.""" + project = _project(tmp_path) + roadmap = project / "blueprint" / "roadmap" + (roadmap / "README.md").write_text( + f"---\nopen_statements: {policy}\n---\n\n# Roadmap\n\n" + "## Definitions\n\n- [Base](base.md)\n\n" + "## Results\n\n- [Open](open.md)\n- [Top](top.md)\n", + encoding="utf-8", + ) + (roadmap / "open.md").write_text( + "---\ndeclaration: theorem\nstatement: formalized\n---\n\n" + "# Open\n\nA statement whose proof is still sorry.\n\n## Depends on\n\n- [Base](base.md)\n", + encoding="utf-8", + ) + top = roadmap / "top.md" + top.write_text( + top.read_text(encoding="utf-8") + "\n## Proof depends on\n\n- [Open](open.md)\n", encoding="utf-8" + ) + return project + + +def test_a_conditional_proof_names_the_open_statements_it_assumes(tmp_path: Path) -> None: + project = _conditional_project(tmp_path, "allowed") + + render_site(project / "blueprint", tmp_path / "out", lean_root=project) + page = (tmp_path / "out/roadmap/README.md").read_text(encoding="utf-8") + top = page[page.index('id="top"'):] + + assert '
conditionally proved' in top + assert ( + 'Assumes' + 'Theorem 1 (Open)' + " (open statements whose Lean proofs are still sorry)" + ) in top + # Only the conditional proof carries the row; the open statement assumes nothing. + assert page.count('Assumes') == 1 + + css = (tmp_path / "out/stylesheets/blueprint.css").read_text(encoding="utf-8") + assert ".bp-ref-conditional::before, .bp-swatch-conditional { background: #E9DFFC; border-color: #6B3FCF; }" in css + assert "[data-md-color-scheme=slate] .bp-ref-conditional::before" in css + assert ".bp-conditional .bp-mark { color: #6B3FCF; }" in css + + +def test_the_strict_policy_shows_the_same_proof_as_proved_with_no_assumptions(tmp_path: Path) -> None: + project = _conditional_project(tmp_path, "forbidden") + + render_site(project / "blueprint", tmp_path / "out", lean_root=project) + page = (tmp_path / "out/roadmap/README.md").read_text(encoding="utf-8") + + assert '
Assumes' not in page diff --git a/tests/test_runtime.py b/tests/test_runtime.py index 247419ea..f67c09ae 100644 --- a/tests/test_runtime.py +++ b/tests/test_runtime.py @@ -284,3 +284,61 @@ def test_adapter_rejects_inconsistent_hand_built_graph_without_host_paths(tmp_pa "chapter/section/base: dependency union does not match typed dependencies", ) assert str(tmp_path) not in str(error.value) + + +def _policy_project(tmp_path: Path, policy: str | None) -> Path: + """`_project` plus an open theorem, a reduction proved from it, and a statement waiting on a gap.""" + project = _project(tmp_path) + if policy is not None: + _article(project, "README.md", title="Roadmap", open_statements=policy) + theorem = {"declaration": "theorem", "statement": "formalized"} + _article(project, "chapter/section/open.md", title="Open", lean="Project.open_thm", **theorem) + _article( + project, + "chapter/section/reduction.md", + title="Reduction", + lean="Project.reduction", + proof="formalized", + proof_dependencies=("open.md",), + **theorem, + ) + _article(project, "chapter/section/gap.md", title="Gap", declaration="theorem") + _article(project, "chapter/section/waiting.md", title="Waiting", proof_dependencies=("gap.md",), **theorem) + return project + + +def _status_payloads(project: Path) -> tuple[bool, dict[str, dict[str, object]]]: + payload = json.loads(load_runtime_graph(project).to_json()) + return payload["open_statements"], { + node["id"].removeprefix("chapter/section/"): node["status"] for node in payload["nodes"] + } + + +def test_strict_runtime_readiness_comes_from_the_derived_status(tmp_path: Path) -> None: + """Runtime readiness once ignored proof prerequisites that `work list` enforced.""" + open_statements, statuses = _status_payloads(_policy_project(tmp_path, None)) + + assert open_statements is False + reduction = statuses["reduction"] + assert (reduction["state"], reduction["can_state"], reduction["can_prove"]) == ("proved", False, False) + assert (reduction["assumes"], reduction["waiting_on"]) == ([], []) + waiting = statuses["waiting"] + assert (waiting["state"], waiting["can_state"], waiting["can_prove"]) == ("stated", False, False) + assert waiting["waiting_on"] == ["chapter/section/gap"] + assert all(status["assumes"] == [] for status in statuses.values()) + + +def test_open_runtime_records_the_policy_and_what_each_proof_assumes(tmp_path: Path) -> None: + open_statements, statuses = _status_payloads(_policy_project(tmp_path, "allowed")) + + assert open_statements is True + reduction = statuses["reduction"] + assert (reduction["state"], reduction["can_state"], reduction["can_prove"]) == ("conditional", True, True) + assert reduction["assumes"] == ["chapter/section/open"] + assert reduction["waiting_on"] == [] + assert not reduction["fully_proved"] + waiting = statuses["waiting"] + assert (waiting["state"], waiting["can_state"], waiting["can_prove"]) == ("stated", True, False) + assert (waiting["assumes"], waiting["waiting_on"]) == ([], ["chapter/section/gap"]) + assert statuses["open"]["state"] == "can_prove" + assert statuses["open"]["assumes"] == [] diff --git a/tests/test_status.py b/tests/test_status.py index 0fad04b4..68ee4195 100644 --- a/tests/test_status.py +++ b/tests/test_status.py @@ -4,8 +4,8 @@ import pytest -from autoform_cli.graph import load_graph -from autoform_cli.status import derive, summarize +from autoform_cli.graph import Graph, Node, load_graph +from autoform_cli.status import NodeStatus, derive, summarize def _node(blueprint: Path, relative: str, body: str = "", **metadata: str) -> None: @@ -172,3 +172,163 @@ def test_assumptions_and_propositions_keep_the_proof_obligation( _node(blueprint, "d.md", declaration=declaration, statement="formalized") assert not derive(load_graph(blueprint))["d"].proved + + +def _mixed_statuses(blueprint: Path, policy: str) -> dict[str, NodeStatus]: + """One graph with every kind of node the two policies treat differently. + + ``hidden``, ``open``, ``inner`` and ``side`` are open statements: stated + theorems with no Lean proof. ``reduction`` and ``top`` are proved on top of + them, ``up`` is in Mathlib, and ``gap`` and ``bridge`` are not stated. + """ + _node(blueprint, "README.md", open_statements=policy) + _node(blueprint, "kind.md", declaration="def", statement="formalized") + theorem = {"declaration": "theorem"} + stated = {**theorem, "statement": "formalized"} + proved = {**stated, "proof": "formalized"} + _node(blueprint, "lemma.md", "## Depends on\n\n- [Kind](kind.md)\n", **proved) + _node(blueprint, "hidden.md", **stated) + _node(blueprint, "up.md", "## Depends on\n\n- [Hidden](hidden.md)\n", **theorem, mathlib="true") + _node(blueprint, "open.md", "## Depends on\n\n- [Kind](kind.md)\n", **stated) + _node( + blueprint, + "reduction.md", + "## Depends on\n\n- [Lemma](lemma.md)\n\n## Proof depends on\n\n- [Open](open.md)\n", + **proved, + ) + _node(blueprint, "gap.md", **theorem) + _node(blueprint, "inner.md", **stated) + _node(blueprint, "side.md", **stated) + _node( + blueprint, + "outer.md", + "## Depends on\n\n- [Inner](inner.md)\n\n## Proof depends on\n\n- [Side](side.md)\n", + **stated, + ) + _node(blueprint, "bridge.md", "## Depends on\n\n- [Open](open.md)\n", **theorem) + _node( + blueprint, + "top.md", + "## Depends on\n\n- [Up](up.md)\n\n## Proof depends on\n\n- [Outer](outer.md)\n- [Bridge](bridge.md)\n", + **proved, + ) + _node( + blueprint, + "waits.md", + "## Depends on\n\n- [Gap](gap.md)\n\n## Proof depends on\n\n- [Gap](gap.md)\n- [Bridge](bridge.md)\n", + **theorem, + ) + _node( + blueprint, + "pending.md", + "## Proof depends on\n\n- [Gap](gap.md)\n- [Reduction](reduction.md)\n", + **stated, + ) + return derive(load_graph(blueprint)) + + +def _readiness( + statuses: dict[str, NodeStatus], +) -> dict[str, tuple[str, bool, bool, tuple[str, ...], tuple[str, ...]]]: + return { + node_id: (status.key, status.can_state, status.can_prove, status.waiting_on, status.assumes) + for node_id, status in statuses.items() + } + + +def test_the_strict_policy_waits_for_proof_prerequisites_to_be_proved(tmp_path: Path) -> None: + statuses = _mixed_statuses(tmp_path / "blueprint", "forbidden") + + assert _readiness(statuses) == { + "roadmap": ("can_state", True, False, (), ()), + "kind": ("fully_proved", True, True, (), ()), + "lemma": ("fully_proved", True, True, (), ()), + "hidden": ("can_prove", True, True, (), ()), + "up": ("mathlib", True, True, (), ()), + "open": ("can_prove", True, True, (), ()), + # Proved, so nothing is waited on, but under this policy its unproved + # proof prerequisite would have kept the statement from landing. + "reduction": ("proved", False, False, (), ()), + "gap": ("can_state", True, False, (), ()), + "inner": ("can_prove", True, True, (), ()), + "side": ("can_prove", True, True, (), ()), + "outer": ("stated", False, False, ("side",), ()), + "bridge": ("can_state", True, False, (), ()), + "top": ("proved", False, False, (), ()), + # Unstated statement prerequisites first, then unproved proof ones, once each. + "waits": ("planned", False, False, ("gap", "bridge"), ()), + "pending": ("stated", False, False, ("gap",), ()), + } + + +def test_the_open_policy_waits_only_for_prerequisites_to_be_stated(tmp_path: Path) -> None: + statuses = _mixed_statuses(tmp_path / "blueprint", "allowed") + + assert _readiness(statuses) == { + "roadmap": ("can_state", True, False, (), ()), + "kind": ("fully_proved", True, True, (), ()), + "lemma": ("fully_proved", True, True, (), ()), + "hidden": ("can_prove", True, True, (), ()), + "up": ("mathlib", True, True, (), ()), + "open": ("can_prove", True, True, (), ()), + "reduction": ("conditional", True, True, (), ("open",)), + "gap": ("can_state", True, False, (), ()), + "inner": ("can_prove", True, True, (), ()), + "side": ("can_prove", True, True, (), ()), + "outer": ("can_prove", True, True, (), ("inner", "side")), + "bridge": ("can_state", True, False, (), ("open",)), + # can_prove reads the prerequisites, not the assertion: bridge is not stated. + "top": ("conditional", True, False, (), ("inner", "outer")), + # An unstated node waits only on its statement prerequisites. + "waits": ("planned", False, False, ("gap",), ()), + "pending": ("stated", True, False, ("gap",), ("open",)), + } + + +@pytest.mark.parametrize("policy", ["forbidden", "allowed"]) +def test_a_fully_proved_node_assumes_nothing(tmp_path: Path, policy: str) -> None: + statuses = _mixed_statuses(tmp_path / "blueprint", policy) + + fully_proved = {node_id for node_id, status in statuses.items() if status.fully_proved} + assert fully_proved == {"kind", "lemma"} + assert all(status.assumes == () for status in statuses.values() if status.fully_proved) + + +def test_assumptions_reach_what_the_lean_proof_can_reach(tmp_path: Path) -> None: + """``top`` rests on ``up``, ``outer`` and ``bridge``, and each stops the walk differently. + + Mathlib ``up`` is upstream, so ``hidden`` behind it is not reached. ``bridge`` + is not stated, so it has no Lean declaration to lead to ``open``. ``outer`` + is open: its own sorry and its statement prerequisite ``inner`` are reached, + but ``side``, which only its missing proof would use, is not. + """ + statuses = _mixed_statuses(tmp_path / "blueprint", "allowed") + + assert statuses["top"].assumes == ("inner", "outer") + # The walk stops at bridge for its dependents, but bridge's own proof would + # still rest on open. A Mathlib node rests on nothing. + assert statuses["bridge"].assumes == ("open",) + assert statuses["up"].assumes == () + + +@pytest.mark.parametrize("open_statements", [False, True]) +def test_a_missing_dependency_counts_as_neither_stated_nor_proved( + tmp_path: Path, open_statements: bool +) -> None: + node = Node( + id="result", + title="Result", + path=tmp_path / "result.md", + dependencies=("ghost", "phantom"), + statement_dependencies=("ghost",), + proof_dependencies=("phantom",), + declaration="theorem", + statement_formalized=True, + ) + graph = Graph(blueprint_dir=tmp_path, nodes={"result": node}, open_statements=open_statements) + + status = derive(graph)["result"] + + assert (status.key, status.can_state, status.can_prove) == ("stated", False, False) + assert status.waiting_on == ("ghost", "phantom") + assert status.assumes == () diff --git a/tests/test_visualization.py b/tests/test_visualization.py index a85dda67..85d6af3c 100644 --- a/tests/test_visualization.py +++ b/tests/test_visualization.py @@ -132,6 +132,27 @@ def test_green_stops_at_an_unproved_prerequisite(tmp_path: Path) -> None: assert statuses["top"].key == "proved" +def test_a_conditional_proof_has_its_own_colour_and_legend_entry(tmp_path: Path) -> None: + """A proof resting on a sorry must not share the fully proved green.""" + blueprint = tmp_path / "blueprint" + _write_node(blueprint / "roadmap" / "README.md", "Roadmap", open_statements="allowed") + _write_node(blueprint / "roadmap" / "open.md", "Open", declaration="theorem", statement="formalized") + _write_node( + blueprint / "roadmap" / "top.md", + "Top", + [("Open", "open.md")], + declaration="theorem", + statement="formalized", + proof="formalized", + ) + + document = export_graph(blueprint).read_text(encoding="utf-8") + + assert '("Top"):::conditional' in document + assert f"classDef conditional fill:{_state('conditional').fill}" in document + assert '' in document + assert "Proof compiles, but rests on an open statement whose Lean proof is still sorry." in document + def test_cli_writes_only_the_graph_by_default( tmp_path: Path, capsys: pytest.CaptureFixture[str] ) -> None: diff --git a/tests/test_work.py b/tests/test_work.py index efc094c1..a13c7310 100644 --- a/tests/test_work.py +++ b/tests/test_work.py @@ -325,19 +325,22 @@ def test_work_cli_emits_stable_json(tmp_path: Path, capsys) -> None: assert cli.main(["work", "list", str(project), "--lean-root", str(project), "--json"]) == 0 frontier = json.loads(capsys.readouterr().out) - assert set(frontier) == {"items", "schema", "source_revision"} + assert set(frontier) == {"items", "open_statements", "schema", "source_revision"} assert frontier["schema"] == WORK_SCHEMA + assert frontier["open_statements"] is False assert frontier["source_revision"] == source_revision assert [item["phase"] for item in frontier["items"]] == ["proof", "statement"] assert frontier["items"][0] == { "article_id": "af_000000000000000000000003", "article_path": "blueprint/roadmap/chapter/prove.md", "article_revision": prove_revision, + "assumes": [], "blockers": [], "claim_target": "af_000000000000000000000003", "dependencies": ["chapter/base"], "lean_targets": [{"declaration": "Project.prove", "source_file": "Project.lean"}], "node_id": "chapter/prove", + "open_statements": False, "phase": "proof", "ready": True, "source_targets": [], @@ -481,3 +484,335 @@ def test_node_ids_cannot_impersonate_article_ids(tmp_path: Path, capsys) -> None assert cli.main(["work", "context", "af_000000000000000000000002", str(project)]) == 2 assert "node id has the form of an article_id" in capsys.readouterr().err + + +def _policy_project(tmp_path: Path, policy: str | None) -> Path: + """`_project` plus an open theorem, a reduction proved from it, and articles resting on them. + + *policy* is the `open_statements` value in `roadmap/README.md`; ``None`` + writes no roadmap page, so the project keeps the default strict policy. + """ + project = _project(tmp_path) + if policy is not None: + (project / "blueprint/roadmap/README.md").write_text( + f"---\nopen_statements: {policy}\n---\n\n# Roadmap\n", encoding="utf-8" + ) + stated = ["declaration: theorem", "statement: formalized"] + _article( + project, + "open.md", + title="Open", + metadata=["article_id: af_00000000000000000000000b", *stated, "lean: Project.open_thm Project.open_aux"], + depends="base.md", + ) + _article( + project, + "reduction.md", + title="Reduction", + metadata=["article_id: af_00000000000000000000000c", *stated, "proof: formalized", "lean: Project.reduction"], + proof_depends="open.md", + ) + _article( + project, + "uses.md", + title="Uses", + metadata=["article_id: af_00000000000000000000000d", *stated, "lean: Project.uses"], + depends="open.md", + ) + _article( + project, + "corollary.md", + title="Corollary", + metadata=["article_id: af_00000000000000000000000e", "declaration: theorem"], + depends="reduction.md", + ) + # Neither belongs in the assumption contract: one is upstream, the other names no declaration. + _article( + project, + "upstream.md", + title="Upstream", + metadata=["declaration: theorem", "mathlib: true", "lean: Project.upstream"], + ) + _article(project, "unnamed.md", title="Unnamed", metadata=["declaration: def", "statement: formalized"]) + return project + + +def _blocked_articles(project: Path) -> None: + """Add articles whose prerequisites block them differently under the two policies.""" + _article( + project, + "waits.md", + title="Waits", + metadata=["article_id: af_000000000000000000000010", "declaration: theorem"], + depends="state.md", + proof_depends="prove.md", + ) + _edit(project, "waits.md", "- [dependency](prove.md)", "- [dependency](prove.md)\n- [dependency](state.md)") + _article( + project, + "stuck.md", + title="Stuck", + metadata=[ + "article_id: af_000000000000000000000011", + "declaration: theorem", + "statement: formalized", + "lean: Project.stuck", + ], + depends="blocked.md", + proof_depends="prove.md", + ) + _edit(project, "stuck.md", "- [dependency](prove.md)", "- [dependency](prove.md)\n- [dependency](state.md)") + _article( + project, + "main.md", + title="Main", + metadata=[ + "article_id: af_000000000000000000000012", + "declaration: theorem", + "statement: formalized", + "lean: Project.main", + ], + proof_depends="prove.md", + ) + + +def _blockers(project: Path) -> dict[str, tuple[str | None, tuple[str, ...]]]: + selectors = ("chapter/waits", "chapter/stuck", "chapter/main") + items = {selector: work_context(project, selector)[1] for selector in selectors} + return {selector: (item.phase, item.blockers) for selector, item in items.items()} + + +def test_strict_blockers_list_unstated_statement_prerequisites_then_unproved_proof_ones( + tmp_path: Path, +) -> None: + project = _project(tmp_path) + _blocked_articles(project) + + assert _blockers(project) == { + "chapter/waits": (None, ("chapter/state", "chapter/prove")), + "chapter/stuck": (None, ("chapter/blocked", "chapter/prove", "chapter/state")), + "chapter/main": (None, ("chapter/prove",)), + } + + +def test_open_blockers_wait_only_for_prerequisites_to_be_stated(tmp_path: Path) -> None: + project = _project(tmp_path) + (project / "blueprint/roadmap/README.md").write_text( + "---\nopen_statements: allowed\n---\n\n# Roadmap\n", encoding="utf-8" + ) + _blocked_articles(project) + + assert _blockers(project) == { + # Unstated, so only its statement prerequisites can hold it back. + "chapter/waits": (None, ("chapter/state",)), + # Stated: prove is stated too, so only the unstated prerequisites remain. + "chapter/stuck": (None, ("chapter/blocked", "chapter/state")), + "chapter/main": ("proof", ()), + } + _, main = work_context(project, "chapter/main") + assert main.assumes == ("chapter/prove",) + assert "chapter/main" in {item.node_id for item in list_ready_work(project).items} + + +def test_strict_work_text_is_unchanged(tmp_path: Path, capsys) -> None: + project = _policy_project(tmp_path, None) + revision = load_runtime_graph(project).source_revision + uses = hashlib.sha256((project / "blueprint/roadmap/chapter/uses.md").read_bytes()).hexdigest() + + assert cli.main(["work", "list", str(project)]) == 0 + assert capsys.readouterr().out == ( + "statement: chapter/corollary [af_00000000000000000000000e] - Corollary\n" + "proof: chapter/open [af_00000000000000000000000b] - Open\n" + "proof: chapter/prove [af_000000000000000000000003] - Prove me\n" + "statement: chapter/state [af_000000000000000000000002] - State me\n" + "proof: chapter/uses [af_00000000000000000000000d] - Uses\n" + ) + + assert cli.main(["work", "context", "chapter/uses", str(project)]) == 0 + assert capsys.readouterr().out == ( + "Uses (chapter/uses)\n" + "State: can_prove\n" + "Phase: proof\n" + "Claim target: af_00000000000000000000000d\n" + "Article: blueprint/roadmap/chapter/uses.md\n" + f"Article revision: {uses}\n" + f"Graph source revision: {revision}\n" + "Dependencies: chapter/open\n" + "Lean: Project.uses\n" + ) + + +def test_open_work_text_names_the_policy_and_what_each_item_assumes(tmp_path: Path, capsys) -> None: + project = _policy_project(tmp_path, "allowed") + revision = load_runtime_graph(project).source_revision + uses = hashlib.sha256((project / "blueprint/roadmap/chapter/uses.md").read_bytes()).hexdigest() + + assert cli.main(["work", "list", str(project)]) == 0 + assert capsys.readouterr().out == ( + "Open statements: allowed (a statement may land with a sorry proof)\n" + "statement: chapter/corollary [af_00000000000000000000000e] - Corollary\n" + " assumes: chapter/open\n" + "proof: chapter/open [af_00000000000000000000000b] - Open\n" + "proof: chapter/prove [af_000000000000000000000003] - Prove me\n" + "statement: chapter/state [af_000000000000000000000002] - State me\n" + "proof: chapter/uses [af_00000000000000000000000d] - Uses\n" + " assumes: chapter/open\n" + ) + + assert cli.main(["work", "context", "chapter/uses", str(project)]) == 0 + assert capsys.readouterr().out == ( + "Uses (chapter/uses)\n" + "State: can_prove\n" + "Phase: proof\n" + "Open statements: allowed\n" + "Assumes: chapter/open\n" + "Claim target: af_00000000000000000000000d\n" + "Article: blueprint/roadmap/chapter/uses.md\n" + f"Article revision: {uses}\n" + f"Graph source revision: {revision}\n" + "Dependencies: chapter/open\n" + "Lean: Project.uses\n" + ) + + assert cli.main(["work", "context", "chapter/prove", str(project)]) == 0 + context = capsys.readouterr().out.splitlines() + assert "Open statements: allowed" in context + assert not any(line.startswith("Assumes:") for line in context) + + assert cli.main(["work", "list", str(project), "--json"]) == 0 + frontier = json.loads(capsys.readouterr().out) + assert frontier["open_statements"] is True + assert {item["node_id"]: item["assumes"] for item in frontier["items"]} == { + "chapter/corollary": ["chapter/open"], + "chapter/open": [], + "chapter/prove": [], + "chapter/state": [], + "chapter/uses": ["chapter/open"], + } + assert all(item["open_statements"] is True for item in frontier["items"]) + + +def _contract_article( + node_id: str, + article_id: str, + state: str, + declarations: list[str], + *, + open_: bool = False, + assumes: tuple[str, ...] = (), + allowed: tuple[str, ...] = (), +) -> dict[str, object]: + return { + "allowed_open_declarations": list(allowed), + "article_id": article_id, + "assumes": list(assumes), + "declarations": declarations, + "id": node_id, + "open": open_, + "state": state, + } + + +def test_work_assumptions_under_the_strict_policy_lists_articles_with_nothing_open( + tmp_path: Path, capsys +) -> None: + project = _policy_project(tmp_path, "forbidden") + + assert cli.main(["work", "assumptions", str(project), "--json"]) == 0 + output = capsys.readouterr().out + contract = json.loads(output) + assert output == json.dumps(contract, sort_keys=True, separators=(",", ":")) + "\n" + assert contract == { + "schema": work_module.ASSUMPTIONS_SCHEMA, + "open_statements": False, + "source_revision": load_runtime_graph(project).source_revision, + "articles": [ + _contract_article("chapter/base", "af_000000000000000000000001", "fully_proved", ["Project.base"]), + _contract_article( + "chapter/open", "af_00000000000000000000000b", "can_prove", ["Project.open_thm", "Project.open_aux"] + ), + _contract_article("chapter/prove", "af_000000000000000000000003", "can_prove", ["Project.prove"]), + _contract_article("chapter/reduction", "af_00000000000000000000000c", "proved", ["Project.reduction"]), + _contract_article("chapter/uses", "af_00000000000000000000000d", "can_prove", ["Project.uses"]), + ], + } + + assert cli.main(["work", "assumptions", str(project)]) == 0 + assert capsys.readouterr().out == "Open statements: forbidden\n" + + +def test_work_assumptions_under_the_open_policy_bounds_each_article( + tmp_path: Path, capsys, monkeypatch: pytest.MonkeyPatch +) -> None: + project = _policy_project(tmp_path, "allowed") + open_declarations = ("Project.open_aux", "Project.open_thm") + + assert cli.main(["work", "assumptions", str(project), "--json"]) == 0 + contract = json.loads(capsys.readouterr().out) + assert contract == { + "schema": "autoform-assumptions/v1", + "open_statements": True, + "source_revision": load_runtime_graph(project).source_revision, + "articles": [ + _contract_article("chapter/base", "af_000000000000000000000001", "fully_proved", ["Project.base"]), + # Declarations keep their authored order; the allowance is sorted. + _contract_article( + "chapter/open", + "af_00000000000000000000000b", + "can_prove", + ["Project.open_thm", "Project.open_aux"], + open_=True, + allowed=open_declarations, + ), + _contract_article( + "chapter/prove", + "af_000000000000000000000003", + "can_prove", + ["Project.prove"], + open_=True, + allowed=("Project.prove",), + ), + _contract_article( + "chapter/reduction", + "af_00000000000000000000000c", + "conditional", + ["Project.reduction"], + assumes=("chapter/open",), + allowed=open_declarations, + ), + _contract_article( + "chapter/uses", + "af_00000000000000000000000d", + "can_prove", + ["Project.uses"], + open_=True, + assumes=("chapter/open",), + allowed=(*open_declarations, "Project.uses"), + ), + ], + } + + monkeypatch.chdir(project) + assert cli.main(["work", "assumptions", "--json"]) == 0 + assert json.loads(capsys.readouterr().out) == contract + + assert cli.main(["work", "assumptions", str(project / "blueprint")]) == 0 + assert capsys.readouterr().out == ( + "Open statements: allowed\n" + "open: chapter/open (Project.open_thm, Project.open_aux)\n" + "open: chapter/prove (Project.prove)\n" + "conditional: chapter/reduction assumes chapter/open\n" + "open: chapter/uses (Project.uses)\n" + "conditional: chapter/uses assumes chapter/open\n" + ) + + +def test_work_assumptions_reports_errors_on_stderr_with_exit_2(tmp_path: Path, capsys) -> None: + assert cli.main(["work", "assumptions", str(tmp_path / "missing")]) == 2 + assert capsys.readouterr().err == "error: project or blueprint directory does not exist\n" + + project = _policy_project(tmp_path, "maybe") + assert cli.main(["work", "assumptions", str(project), "--json"]) == 2 + captured = capsys.readouterr() + assert captured.out == "" + assert captured.err == "error: roadmap:2: 'open_statements' accepts allowed or forbidden\n" From ce0c1235ef3e6a7f5155f0c47f5bdde3a2ab4300 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 05:31:52 -0400 Subject: [PATCH 02/38] Pin readiness wording, Assumes row order and the empty open frontier Nothing pinned the reworded legend meanings for can_prove and can_state or the policy-neutral Next up explanations, so add a legend test and a landing-page test for both phases. The conditional article test now checks that the Assumes row sits after the Lean row and before Discussion, as spec A3 places it. Under the open policy, `work list` names the policy even when nothing is ready; a test covers that branch. Also restore the two blank lines after the conditional legend test. --- tests/test_render.py | 25 +++++++++++++++++++++++++ tests/test_visualization.py | 15 +++++++++++++++ tests/test_work.py | 14 ++++++++++++++ 3 files changed, 54 insertions(+) diff --git a/tests/test_render.py b/tests/test_render.py index ae5cfea8..cee93485 100644 --- a/tests/test_render.py +++ b/tests/test_render.py @@ -1315,6 +1315,9 @@ def test_a_conditional_proof_names_the_open_statements_it_assumes(tmp_path: Path ) in top # Only the conditional proof carries the row; the open statement assumes nothing. assert page.count('Assumes') == 1 + # The row follows the implementation row and precedes Discussion. + meta = top[top.index('
'):top.index('
')] + assert re.findall(r'([^<]+)', meta) == ["Lean", "Assumes", "Discussion"] css = (tmp_path / "out/stylesheets/blueprint.css").read_text(encoding="utf-8") assert ".bp-ref-conditional::before, .bp-swatch-conditional { background: #E9DFFC; border-color: #6B3FCF; }" in css @@ -1330,3 +1333,25 @@ def test_the_strict_policy_shows_the_same_proof_as_proved_with_no_assumptions(tm assert '
Assumes' not in page + + +@pytest.mark.parametrize( + ("dropped", "why"), + [ + ("proof: formalized\n", "Its prerequisites are ready, so the proof can be written now."), + ( + "statement: formalized\nproof: formalized\n", + "Its prerequisites are ready, so the statement can be written down.", + ), + ], +) +def test_next_up_explains_readiness_without_naming_a_policy(tmp_path: Path, dropped: str, why: str) -> None: + project = _project(tmp_path) + top = project / "blueprint" / "roadmap" / "top.md" + top.write_text(top.read_text(encoding="utf-8").replace(dropped, ""), encoding="utf-8") + + render_site(project / "blueprint", tmp_path / "out", lean_root=project) + landing = (tmp_path / "out/README.md").read_text(encoding="utf-8") + + assert '
' in landing + assert f'
{why}
' in landing diff --git a/tests/test_visualization.py b/tests/test_visualization.py index 85d6af3c..2d991aca 100644 --- a/tests/test_visualization.py +++ b/tests/test_visualization.py @@ -153,6 +153,21 @@ def test_a_conditional_proof_has_its_own_colour_and_legend_entry(tmp_path: Path) assert '' in document assert "Proof compiles, but rests on an open statement whose Lean proof is still sorry." in document + +def test_the_legend_explains_readiness_without_naming_a_policy(tmp_path: Path) -> None: + """Readiness depends on the project's policy, so the legend words it without naming one.""" + blueprint = tmp_path / "blueprint" + _write_node(blueprint / "roadmap" / "stated.md", "Stated", declaration="theorem", statement="formalized") + _write_node(blueprint / "roadmap" / "unstated.md", "Unstated", declaration="theorem") + statuses = derive(load_graph(blueprint)) + + legend = mermaid.render_legend(statuses) + + assert [statuses[key].key for key in ("stated", "unstated")] == ["can_prove", "can_state"] + assert "Statement is in Lean and nothing it needs is blocked, so the proof can start." in legend + assert "Nothing it needs is blocked, so the statement can be written in Lean." in legend + + def test_cli_writes_only_the_graph_by_default( tmp_path: Path, capsys: pytest.CaptureFixture[str] ) -> None: diff --git a/tests/test_work.py b/tests/test_work.py index a13c7310..8af35491 100644 --- a/tests/test_work.py +++ b/tests/test_work.py @@ -692,6 +692,20 @@ def test_open_work_text_names_the_policy_and_what_each_item_assumes(tmp_path: Pa assert all(item["open_statements"] is True for item in frontier["items"]) +def test_open_work_list_names_the_policy_even_when_nothing_is_ready(tmp_path: Path, capsys) -> None: + project = tmp_path / "project" + _article(project, "README.md", title="Chapter", metadata=[]) + (project / "blueprint/roadmap/README.md").write_text( + "---\nopen_statements: allowed\n---\n\n# Roadmap\n", encoding="utf-8" + ) + + assert cli.main(["work", "list", str(project)]) == 0 + assert capsys.readouterr().out == ( + "Open statements: allowed (a statement may land with a sorry proof)\n" + "No ready formalization work.\n" + ) + + def _contract_article( node_id: str, article_id: str, From 9e29e7873c7a20ed3212f4cc55f552848080e6cd Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 05:33:50 -0400 Subject: [PATCH 03/38] Cover the unreadable-path and output-error branches of work assumptions `work assumptions` maps OSError and RuntimeError to a path-free message with exit 2 and leaves output errors to propagate, like `work list`, but only its validation errors were tested. Extend the existing error tests to pin both branches for the new command. --- tests/test_work.py | 18 +++++++++++++++++- 1 file changed, 17 insertions(+), 1 deletion(-) diff --git a/tests/test_work.py b/tests/test_work.py index 8af35491..916037ba 100644 --- a/tests/test_work.py +++ b/tests/test_work.py @@ -474,6 +474,8 @@ def write(self, text: str) -> int: monkeypatch.setattr(sys, "stdout", ClosedPipe()) with pytest.raises(BrokenPipeError): cli.main(["work", "list", str(project)]) + with pytest.raises(BrokenPipeError): + cli.main(["work", "assumptions", str(project)]) def test_node_ids_cannot_impersonate_article_ids(tmp_path: Path, capsys) -> None: @@ -821,7 +823,9 @@ def test_work_assumptions_under_the_open_policy_bounds_each_article( ) -def test_work_assumptions_reports_errors_on_stderr_with_exit_2(tmp_path: Path, capsys) -> None: +def test_work_assumptions_reports_errors_on_stderr_with_exit_2( + tmp_path: Path, capsys, monkeypatch: pytest.MonkeyPatch +) -> None: assert cli.main(["work", "assumptions", str(tmp_path / "missing")]) == 2 assert capsys.readouterr().err == "error: project or blueprint directory does not exist\n" @@ -830,3 +834,15 @@ def test_work_assumptions_reports_errors_on_stderr_with_exit_2(tmp_path: Path, c captured = capsys.readouterr() assert captured.out == "" assert captured.err == "error: roadmap:2: 'open_statements' accepts allowed or forbidden\n" + + for failure in ( + PermissionError(13, "Permission denied", "/private/secret/blueprint"), + RuntimeError("Symlink loop from '/private/secret/blueprint'"), + ): + + def unreadable(*_args, failure: Exception = failure, **_kwargs): + raise failure + + monkeypatch.setattr(cli, "assumption_contract", unreadable) + assert cli.main(["work", "assumptions", str(project)]) == 2 + assert capsys.readouterr().err == "error: project or blueprint path cannot be read\n" From ac7be46adbccdfa0d095f70490333e8016d6cd69 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 05:25:30 -0400 Subject: [PATCH 04/38] Add all-or-nothing multi-key claims ClaimBoard gains acquire_many, renew_many, and release_many. Each reads every key with one ls-remote, applies the per-key checks of the single-key method, and changes all refs in one git push --atomic with a --force-with-lease per ref, so a batch never holds some claims and not others. A lost race or a held claim returns a ClaimBatchResult naming the blocking keys, read from the porcelain status lines or the remote's "cannot lock ref" error. A board that does not support atomic pushes raises ClaimTransportError instead of falling back to separate pushes. autoform claim acquire|renew|release now takes one or more nodes. One node keeps today's code path and output; several print one line per target on success, one error line naming the blockers on failure, and exit 2 on a duplicate target. --- autoform_cli/__main__.py | 38 +++++- autoform_cli/claims.py | 185 +++++++++++++++++++++++---- tests/test_claim_cli.py | 97 ++++++++++++++- tests/test_claims.py | 263 +++++++++++++++++++++++++++++++++++++++ 4 files changed, 554 insertions(+), 29 deletions(-) diff --git a/autoform_cli/__main__.py b/autoform_cli/__main__.py index 722ba1aa..895c5fe4 100644 --- a/autoform_cli/__main__.py +++ b/autoform_cli/__main__.py @@ -143,7 +143,7 @@ def main(argv: Sequence[str] | None = None) -> int: claim_subparsers = claim.add_subparsers(dest="claim_command", required=True) for operation in ("acquire", "renew", "release"): command = claim_subparsers.add_parser(operation) - command.add_argument("node_id") + command.add_argument("node_id", nargs="+", help="claim target(s); several change all-or-nothing") _add_claim_board_arguments(command) if operation in {"acquire", "renew"}: command.add_argument("--ttl", type=int, default=CLAIM_TTL_S) @@ -585,7 +585,36 @@ def _claim(args: argparse.Namespace) -> int: print(f"removed {board.cleanup()} expired claim(s)") return 0 - key = author_claim_key(args.node_id) + past_tense = {"acquire": "acquired", "renew": "renewed", "release": "released"} + if len(args.node_id) > 1: + # Several targets change in one atomic push, so a failure holds none of them. + targets: dict[str, str] = {} + for node_id in args.node_id: + key = author_claim_key(node_id) + if key in targets: + print(f"error: duplicate claim target: {node_id}", file=sys.stderr) + return 2 + targets[key] = node_id + if operation == "acquire": + result = board.acquire_many(list(targets), ttl=args.ttl, note=args.note) + elif operation == "renew": + result = board.renew_many(list(targets), ttl=args.ttl) + else: + result = board.release_many(list(targets)) + if result: + for key, node_id in targets.items(): + print(f"{past_tense[operation]} {node_id} ({key})") + return 0 + blocking = ", ".join(targets.get(key, key) for key in result.blocking) + reason = f"{result.reason}: {blocking}" if blocking else result.reason + print( + f"error: could not {operation} {', '.join(args.node_id)}; " + f"no claim was {past_tense[operation]}: {reason}" + ) + return 1 + + node_id = args.node_id[0] + key = author_claim_key(node_id) if operation == "acquire": succeeded = board.acquire(key, ttl=args.ttl, note=args.note) elif operation == "renew": @@ -593,10 +622,9 @@ def _claim(args: argparse.Namespace) -> int: else: succeeded = board.release(key) if succeeded: - past_tense = {"acquire": "acquired", "renew": "renewed", "release": "released"} - print(f"{past_tense[operation]} {args.node_id} ({key})") + print(f"{past_tense[operation]} {node_id} ({key})") return 0 - print(f"error: could not {operation} {args.node_id}; ownership is held or unverifiable") + print(f"error: could not {operation} {node_id}; ownership is held or unverifiable") return 1 except (ClaimTransportError, ValueError) as exc: print(f"error: {exc}") diff --git a/autoform_cli/claims.py b/autoform_cli/claims.py index 09901325..d6c44365 100644 --- a/autoform_cli/claims.py +++ b/autoform_cli/claims.py @@ -17,8 +17,9 @@ import subprocess import threading import time +from dataclasses import dataclass from pathlib import Path -from typing import Any, Mapping +from typing import Any, Mapping, Sequence CLAIM_REF_PREFIX = "refs/autoform-claims/" CLAIM_SCHEMA = "autoform-claim/v1" @@ -38,6 +39,9 @@ "remote ref updated since checkout", "cannot lock ref", ) +# When the remote aborts an atomic push, every ref reports "atomic transaction +# failed"; only this error line names the ref whose expected value was gone. +_LOCK_FAILURE_RE = re.compile(r"cannot lock ref '([^']+)'") class ClaimTransportError(RuntimeError): @@ -48,6 +52,22 @@ class MalformedLeaseError(ClaimTransportError): """A claim ref exists, but its lease cannot be verified safely.""" +@dataclass(frozen=True, slots=True) +class ClaimBatchResult: + """The outcome of an all-or-nothing multi-key claim operation. + + It is true exactly when every key changed. Otherwise no key changed, + ``reason`` says why, and ``blocking`` names the keys responsible when known. + """ + + ok: bool + blocking: tuple[str, ...] = () + reason: str = "" + + def __bool__(self) -> bool: + return self.ok + + def _validate_key(key: str) -> str: if not isinstance(key, str) or not CLAIM_KEY_RE.fullmatch(key) or ".." in key: raise ValueError(f"invalid claim key {key!r}") @@ -57,6 +77,32 @@ def _validate_key(key: str) -> str: return key +def _validate_keys(keys: Sequence[str]) -> tuple[str, ...]: + if isinstance(keys, str): + raise TypeError("claim keys must be a sequence of keys, not one string") + batch = tuple(_validate_key(key) for key in keys) + if not batch: + raise ValueError("at least one claim key is required") + seen: set[str] = set() + for key in batch: + if key in seen: + raise ValueError(f"duplicate claim key {key!r}") + seen.add(key) + return batch + + +def _rejected_keys(detail: str, keys_by_ref: Mapping[str, str]) -> tuple[str, ...]: + """Return the keys a refused push reports as stale or contended, in push order.""" + refs = set(_LOCK_FAILURE_RE.findall(detail)) + for line in detail.splitlines(): + # Porcelain status lines read "\t:\t". + flag, _, rest = line.partition("\t") + refspec, _, summary = rest.partition("\t") + if flag == "!" and any(marker in summary.lower() for marker in _CAS_REJECTIONS): + refs.add(refspec.rpartition(":")[2]) + return tuple(key for ref, key in keys_by_ref.items() if ref in refs) + + def _is_finite_number(value: object) -> bool: if isinstance(value, bool): return False @@ -145,6 +191,19 @@ def _remote_oid(self, key: str) -> str | None: line = proc.stdout.strip() return line.split("\t", 1)[0] if line else None + def _remote_oids(self, keys: Sequence[str]) -> dict[str, str]: + """Read every present key's object ID with one ``ls-remote``.""" + keys_by_ref = {self._ref(key): key for key in keys} + proc = self._git(["ls-remote", self.repo_url, *keys_by_ref]) + oids: dict[str, str] = {} + for line in proc.stdout.splitlines(): + oid, separator, ref = line.partition("\t") + # ls-remote patterns also match longer names ending in the same + # path, such as refs/heads/; only the exact ref is the claim. + if separator and ref in keys_by_ref: + oids[keys_by_ref[ref]] = oid + return oids + def _read_lease(self, key: str, oid: str) -> dict[str, Any]: ref = self._ref(key) if self._git(["cat-file", "-e", f"{oid}^{{commit}}"], check=False).returncode != 0: @@ -182,6 +241,27 @@ def _lease_is_valid(lease: Mapping[str, Any], key: str | None = None) -> bool: ) return bool(valid and (key is None or lease.get("resource") == key)) + def _held_by_peer(self, key: str, old: str | None) -> bool: + """Return whether ``old`` is another worker's verified live lease, which blocks acquiring.""" + if old is None: + return False + lease = self._read_lease(key, old) + return bool( + lease is not None + and self._lease_is_valid(lease, key) + and lease.get("owner") != self.worker_id + and not self.expired(lease) + ) + + def _owned_lease(self, key: str, old: str | None) -> dict[str, Any] | None: + """Return this worker's verified lease at ``old``, or ``None`` if absent or not owned.""" + if old is None: + return None + lease = self._read_lease(key, old) + if lease is None or not self._lease_is_valid(lease, key) or lease.get("owner") != self.worker_id: + return None + return lease + def _make_lease_commit(self, key: str, ttl: int | float, note: str = "") -> str: key = _validate_key(key) ttl = _validate_ttl(ttl) @@ -210,24 +290,41 @@ def _make_lease_commit(self, key: str, ttl: int | float, note: str = "") -> str: return self._git(["commit-tree", tree, "-m", message]).stdout.strip() def _cas_push(self, key: str, old: str | None, new: str) -> bool: - ref = self._ref(key) - source = new if new else "" + return self._cas_push_refs([(key, old, new)]).ok + + def _cas_push_refs( + self, + updates: Sequence[tuple[str, str | None, str]], + *, + atomic: bool = False, + ) -> ClaimBatchResult: + """Push ``(key, old, new)`` updates, each leased on its observed ``old``; empty ``new`` deletes. + + With ``atomic`` the remote applies every update or none. A lost + compare-and-swap is a failed result; anything else raises. + """ + keys_by_ref = {self._ref(key): key for key, _, _ in updates} proc = self._git( [ "push", "--quiet", "--porcelain", - f"--force-with-lease={ref}:{old or ''}", + *(["--atomic"] if atomic else []), + *(f"--force-with-lease={self._ref(key)}:{old or ''}" for key, old, _ in updates), self.repo_url, - f"{source}:{ref}", + *(f"{new if new else ''}:{self._ref(key)}" for key, _, new in updates), ], check=False, ) if proc.returncode == 0: - return True + return ClaimBatchResult(True) detail = f"{proc.stdout}\n{proc.stderr}".strip() + if atomic and "does not support --atomic" in detail: + # Falling back to separate pushes could leave a partial claim set. + raise ClaimTransportError("claim board does not support atomic pushes; refusing a multi-key update") if any(marker in detail.lower() for marker in _CAS_REJECTIONS): - return False + stale = _rejected_keys(detail, keys_by_ref) + return ClaimBatchResult(False, stale, "changed concurrently" if stale else "a claim changed concurrently") raise ClaimTransportError(f"claim CAS push failed: {detail[:300]}") def read(self, key: str) -> dict[str, Any] | None: @@ -253,16 +350,8 @@ def acquire(self, key: str, ttl: int | float = CLAIM_TTL_S, steal: bool = False, _validate_ttl(ttl) self._ensure_scratch() old = self._remote_oid(key) - if old is not None: - lease = self._read_lease(key, old) - if ( - lease is not None - and self._lease_is_valid(lease, key) - and lease.get("owner") != self.worker_id - and not self.expired(lease) - and not steal - ): - return False + if self._held_by_peer(key, old) and not steal: + return False new = self._make_lease_commit(key, ttl, note) return self._cas_push(key, old, new) @@ -272,10 +361,8 @@ def renew(self, key: str, ttl: int | float = CLAIM_TTL_S) -> bool: _validate_ttl(ttl) self._ensure_scratch() old = self._remote_oid(key) - if old is None: - return False - lease = self._read_lease(key, old) - if lease is None or not self._lease_is_valid(lease, key) or lease.get("owner") != self.worker_id: + lease = self._owned_lease(key, old) + if lease is None: return False new = self._make_lease_commit(key, ttl, str(lease.get("note", ""))) return self._cas_push(key, old, new) @@ -287,11 +374,62 @@ def release(self, key: str) -> bool: old = self._remote_oid(key) if old is None: return True - lease = self._read_lease(key, old) - if lease is None or not self._lease_is_valid(lease, key) or lease.get("owner") != self.worker_id: + if self._owned_lease(key, old) is None: return False return self._cas_push(key, old, "") + # The *_many methods read every key with one ls-remote, apply the per-key + # checks of the single-key method, and change all keys in one atomic push, + # so a batch never leaves some of its claims taken and others not. + + def acquire_many( + self, + keys: Sequence[str], + *, + ttl: int | float = CLAIM_TTL_S, + note: str = "", + ) -> ClaimBatchResult: + """Acquire every key or none; a live lease of another worker on any key blocks all.""" + keys = _validate_keys(keys) + _validate_ttl(ttl) + self._ensure_scratch() + olds = self._remote_oids(keys) + held = tuple(key for key in keys if self._held_by_peer(key, olds.get(key))) + if held: + return ClaimBatchResult(False, held, "held by another worker") + updates = [(key, olds.get(key), self._make_lease_commit(key, ttl, note)) for key in keys] + return self._cas_push_refs(updates, atomic=True) + + def renew_many(self, keys: Sequence[str], *, ttl: int | float = CLAIM_TTL_S) -> ClaimBatchResult: + """Renew every key or none; each must be this worker's lease, as :meth:`renew` requires.""" + keys = _validate_keys(keys) + _validate_ttl(ttl) + self._ensure_scratch() + olds = self._remote_oids(keys) + notes: dict[str, str] = {} + for key in keys: + lease = self._owned_lease(key, olds.get(key)) + if lease is not None: + notes[key] = str(lease.get("note", "")) + lost = tuple(key for key in keys if key not in notes) + if lost: + return ClaimBatchResult(False, lost, "not held by this worker") + updates = [(key, olds[key], self._make_lease_commit(key, ttl, notes[key])) for key in keys] + return self._cas_push_refs(updates, atomic=True) + + def release_many(self, keys: Sequence[str]) -> ClaimBatchResult: + """Delete every present key or none; absent keys count as released, as in :meth:`release`.""" + keys = _validate_keys(keys) + self._ensure_scratch() + olds = self._remote_oids(keys) + present = [key for key in keys if key in olds] + foreign = tuple(key for key in present if self._owned_lease(key, olds[key]) is None) + if foreign: + return ClaimBatchResult(False, foreign, "not held by this worker") + if not present: + return ClaimBatchResult(True) + return self._cas_push_refs([(key, olds[key], "") for key in present], atomic=True) + def holds(self, key: str) -> bool: """Return whether this worker verifiably owns the current live lease.""" lease = self.read(key) @@ -423,6 +561,7 @@ def _run(self) -> None: "CLAIM_REF_PREFIX", "CLAIM_SCHEMA", "CLAIM_TTL_S", + "ClaimBatchResult", "ClaimBoard", "ClaimTransportError", "Heartbeat", diff --git a/tests/test_claim_cli.py b/tests/test_claim_cli.py index e0a48236..d108066d 100644 --- a/tests/test_claim_cli.py +++ b/tests/test_claim_cli.py @@ -5,7 +5,7 @@ from pathlib import Path from autoform_cli.__main__ import main -from autoform_cli.claims import CLAIM_REF_PREFIX, author_claim_key +from autoform_cli.claims import CLAIM_REF_PREFIX, ClaimBoard, author_claim_key def _bare_repo(tmp_path: Path) -> Path: @@ -102,3 +102,98 @@ def test_claim_cli_refuses_malformed_remote_lease(tmp_path: Path, capsys) -> Non assert main(_args(repo, tmp_path / "scratch", "acquire", node_id)) == 1 assert "invalid lease JSON" in capsys.readouterr().out + + +def _claim_refs(repo: Path) -> list[str]: + listing = subprocess.run( + ["git", "for-each-ref", "--format=%(refname)", CLAIM_REF_PREFIX], + cwd=repo, + capture_output=True, + text=True, + check=True, + ).stdout + return sorted(listing.split()) + + +def _peer_args(repo: Path, scratch: Path, *command: str) -> list[str]: + return ["claim", *command, "--repo", str(repo), "--worker-id", "worker-b", "--scratch", str(scratch)] + + +def test_claim_cli_single_node_output_is_unchanged(tmp_path: Path, capsys, monkeypatch) -> None: + def batch_path(*args: object, **kwargs: object) -> None: + raise AssertionError("one target must take the single-key path") + + for name in ("acquire_many", "renew_many", "release_many"): + monkeypatch.setattr(ClaimBoard, name, batch_path) + repo = _bare_repo(tmp_path) + scratch = tmp_path / "scratch" + key = author_claim_key("node") + + assert main(_args(repo, scratch, "acquire", "node", "--ttl", "600")) == 0 + assert capsys.readouterr().out == f"acquired node ({key})\n" + assert main(_args(repo, scratch, "renew", "node", "--ttl", "600")) == 0 + assert capsys.readouterr().out == f"renewed node ({key})\n" + assert main(_peer_args(repo, tmp_path / "peer", "acquire", "node")) == 1 + assert capsys.readouterr().out == "error: could not acquire node; ownership is held or unverifiable\n" + assert main(_args(repo, scratch, "release", "node")) == 0 + assert capsys.readouterr().out == f"released node ({key})\n" + + +def test_claim_cli_multi_node_round_trip_prints_one_line_per_target(tmp_path: Path, capsys) -> None: + repo = _bare_repo(tmp_path) + scratch = tmp_path / "scratch" + nodes = ["chapter/main theorem", "chapter/helper"] + + def expected(past_tense: str) -> str: + return "".join(f"{past_tense} {node} ({author_claim_key(node)})\n" for node in nodes) + + assert main(_args(repo, scratch, "acquire", *nodes, "--ttl", "600", "--note", "revision")) == 0 + assert capsys.readouterr().out == expected("acquired") + assert main(_args(repo, scratch, "renew", *nodes, "--ttl", "600")) == 0 + assert capsys.readouterr().out == expected("renewed") + assert main(_args(repo, scratch, "list")) == 0 + leases = json.loads(capsys.readouterr().out) + assert sorted(lease["_key"] for lease in leases) == sorted(author_claim_key(node) for node in nodes) + assert {(lease["owner"], lease["note"]) for lease in leases} == {("worker-a", "revision")} + assert main(_args(repo, scratch, "release", *nodes)) == 0 + assert capsys.readouterr().out == expected("released") + assert _claim_refs(repo) == [] + + +def test_claim_cli_multi_node_failure_changes_no_claim(tmp_path: Path, capsys) -> None: + repo = _bare_repo(tmp_path) + scratch = tmp_path / "scratch" + assert main(_peer_args(repo, tmp_path / "peer", "acquire", "b")) == 0 + capsys.readouterr() + + assert main(_args(repo, scratch, "acquire", "a", "b", "c")) == 1 + captured = capsys.readouterr() + assert captured.out == "error: could not acquire a, b, c; no claim was acquired: held by another worker: b\n" + assert captured.err == "" + assert _claim_refs(repo) == [CLAIM_REF_PREFIX + author_claim_key("b")] + + assert main(_args(repo, scratch, "acquire", "a")) == 0 + capsys.readouterr() + assert main(_args(repo, scratch, "release", "a", "b")) == 1 + assert capsys.readouterr().out == ( + "error: could not release a, b; no claim was released: not held by this worker: b\n" + ) + assert _claim_refs(repo) == sorted(CLAIM_REF_PREFIX + author_claim_key(node) for node in ("a", "b")) + + +def test_claim_cli_rejects_duplicate_targets_before_touching_the_board(tmp_path: Path, capsys) -> None: + repo = _bare_repo(tmp_path) + scratch = tmp_path / "scratch" + + assert main(_args(repo, scratch, "acquire", "a", "b", "a")) == 2 + captured = capsys.readouterr() + assert captured.err == "error: duplicate claim target: a\n" + assert captured.out == "" + assert _claim_refs(repo) == [] + assert not scratch.exists() + + +def test_claim_cli_multi_node_transport_failure_is_nonzero(tmp_path: Path, capsys) -> None: + missing = tmp_path / "missing" / "claims.git" + assert main(_args(missing, tmp_path / "scratch", "acquire", "a", "b")) == 1 + assert capsys.readouterr().out.startswith("error: ") diff --git a/tests/test_claims.py b/tests/test_claims.py index 7f140c2d..99fd5fdf 100644 --- a/tests/test_claims.py +++ b/tests/test_claims.py @@ -6,6 +6,7 @@ import os import subprocess import threading +from collections.abc import Sequence from pathlib import Path import pytest @@ -456,3 +457,265 @@ def test_acquire_rejects_nonfinite_expiry_before_commit_or_push( board.acquire("bad-expiry", ttl=1e308) assert _git("for-each-ref", "--format=%(refname)", claims.CLAIM_REF_PREFIX + "bad-expiry", cwd=board_repo) == "" + + +def _refs(repo: Path) -> dict[str, str]: + listing = _git("for-each-ref", "--format=%(refname) %(objectname)", cwd=repo) + return dict(line.split(" ", 1) for line in listing.splitlines()) + + +def test_acquire_many_is_all_or_nothing_when_a_peer_holds_one_key(tmp_path: Path, board_repo: Path) -> None: + peer = _board(tmp_path, board_repo, "worker-b") + assert peer.acquire("held", ttl=600) + before = _refs(board_repo) + board = _board(tmp_path, board_repo, "worker-a") + + result = board.acquire_many(["first", "held", "last"], ttl=600) + + assert result == claims.ClaimBatchResult(False, ("held",), "held by another worker") + assert not result + assert _refs(board_repo) == before + + +def test_acquire_many_takes_over_an_expired_key(tmp_path: Path, board_repo: Path) -> None: + _plant_lease(board_repo, "expired", owner="worker-b") + board = _board(tmp_path, board_repo, "worker-a") + + result = board.acquire_many(["fresh", "expired"], ttl=600, note="revision") + + assert result == claims.ClaimBatchResult(True) + for key in ("fresh", "expired"): + lease = board.read(key) + assert lease["owner"] == "worker-a" + assert lease["note"] == "revision" + assert board.holds(key) + + +def test_acquire_many_changes_no_ref_when_one_lease_goes_stale_before_the_push( + tmp_path: Path, board_repo: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + _plant_lease(board_repo, "contested", owner="worker-b") + board = _board(tmp_path, board_repo, "worker-a") + peer = _board(tmp_path, board_repo, "worker-b") + push = board._cas_push_refs + observed: dict[str, dict[str, str]] = {} + + def peer_renews_first(updates: Sequence[tuple[str, str | None, str]], *, atomic: bool = False): + # The expired lease passed the ownership check; its owner renews it + # before the push, so exactly one of the two leases is stale. + assert peer.renew("contested", ttl=600) + observed["refs"] = _refs(board_repo) + return push(updates, atomic=atomic) + + monkeypatch.setattr(board, "_cas_push_refs", peer_renews_first) + result = board.acquire_many(["free", "contested"], ttl=600) + + assert result == claims.ClaimBatchResult(False, ("contested",), "changed concurrently") + assert _refs(board_repo) == observed["refs"] + assert claims.CLAIM_REF_PREFIX + "free" not in observed["refs"] + assert peer.holds("contested") + + +def test_acquire_many_changes_no_ref_when_the_remote_aborts_the_transaction( + tmp_path: Path, board_repo: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + board = _board(tmp_path, board_repo, "worker-a") + board._ensure_scratch() + rival = _git("commit-tree", _git("mktree", cwd=board_repo, input_text=""), "-m", "rival", cwd=board_repo) + contested = claims.CLAIM_REF_PREFIX + "contested" + # Git runs pre-push after checking every lease against the remote's + # advertisement, so this rival claim meets the remote's own check inside + # the ref transaction instead. + hooks = board.scratch / "hooks" + hooks.mkdir(exist_ok=True) + hook = hooks / "pre-push" + hook.write_text(f'#!/bin/sh\ncat >/dev/null\nexec git --git-dir="{board_repo}" update-ref {contested} {rival}\n') + hook.chmod(0o755) + monkeypatch.setenv("GIT_CONFIG_COUNT", "1") + monkeypatch.setenv("GIT_CONFIG_KEY_0", "core.hooksPath") + monkeypatch.setenv("GIT_CONFIG_VALUE_0", str(hooks)) + + result = board.acquire_many(["free", "contested"], ttl=600) + + assert result == claims.ClaimBatchResult(False, ("contested",), "changed concurrently") + assert _refs(board_repo) == {contested: rival} + + +def test_overlapping_batch_race_has_one_winner_and_no_partial_loser(tmp_path: Path, board_repo: Path) -> None: + boards = {owner: _board(tmp_path, board_repo, owner) for owner in ("worker-a", "worker-b")} + barrier = threading.Barrier(2) + original_remote_oids = claims.ClaimBoard._remote_oids + + def synchronized_remote_oids(self: claims.ClaimBoard, keys: Sequence[str]) -> dict[str, str]: + oids = original_remote_oids(self, keys) + barrier.wait(timeout=60) + return oids + + for board in boards.values(): + board._remote_oids = synchronized_remote_oids.__get__(board, claims.ClaimBoard) # type: ignore[method-assign] + + results: dict[str, claims.ClaimBatchResult] = {} + errors: list[BaseException] = [] + + def acquire(owner: str) -> None: + try: + results[owner] = boards[owner].acquire_many([f"only-{owner}", "shared"], ttl=600) + except BaseException as exc: # pragma: no cover - reported below + errors.append(exc) + + threads = [threading.Thread(target=acquire, args=(owner,)) for owner in boards] + for thread in threads: + thread.start() + for thread in threads: + thread.join(timeout=120) + + assert not errors + assert not any(thread.is_alive() for thread in threads) + winners = [owner for owner, result in results.items() if result] + assert len(winners) == 1 + winner = winners[0] + loser = next(owner for owner in boards if owner != winner) + assert results[loser] == claims.ClaimBatchResult(False, ("shared",), "changed concurrently") + assert sorted(_refs(board_repo)) == [claims.CLAIM_REF_PREFIX + key for key in (f"only-{winner}", "shared")] + assert boards[loser].read("shared")["owner"] == winner + + +def test_renew_many_renews_every_owned_key_and_keeps_notes( + tmp_path: Path, board_repo: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + now = 1_000.0 + monkeypatch.setattr(claims.time, "time", lambda: now) + board = _board(tmp_path, board_repo, "worker-a") + assert board.acquire_many(["first", "second"], ttl=60, note="revision") + monkeypatch.setattr(claims.time, "time", lambda: now + 30) + + assert board.renew_many(["first", "second"], ttl=600) == claims.ClaimBatchResult(True) + for key in ("first", "second"): + lease = board.read(key) + assert lease["expires_at"] == now + 30 + 600 + assert lease["note"] == "revision" + + +def test_renew_many_renews_nothing_unless_every_key_is_owned(tmp_path: Path, board_repo: Path) -> None: + board = _board(tmp_path, board_repo, "worker-a") + peer = _board(tmp_path, board_repo, "worker-b") + assert board.acquire("mine", ttl=600) + assert peer.acquire("theirs", ttl=600) + before = _refs(board_repo) + + result = board.renew_many(["mine", "theirs", "absent"], ttl=900) + + assert result == claims.ClaimBatchResult(False, ("theirs", "absent"), "not held by this worker") + assert _refs(board_repo) == before + + +def test_release_many_deletes_owned_keys_and_treats_absent_keys_as_released(tmp_path: Path, board_repo: Path) -> None: + board = _board(tmp_path, board_repo, "worker-a") + assert board.acquire_many(["first", "second"], ttl=600) + + assert board.release_many(["first", "absent", "second"]) == claims.ClaimBatchResult(True) + assert _refs(board_repo) == {} + assert board.release_many(["first", "second"]) == claims.ClaimBatchResult(True) + + +def test_release_many_releases_nothing_when_one_key_is_foreign(tmp_path: Path, board_repo: Path) -> None: + board = _board(tmp_path, board_repo, "worker-a") + peer = _board(tmp_path, board_repo, "worker-b") + assert board.acquire("mine", ttl=600) + assert peer.acquire("theirs", ttl=600) + before = _refs(board_repo) + + result = board.release_many(["mine", "theirs"]) + + assert result == claims.ClaimBatchResult(False, ("theirs",), "not held by this worker") + assert _refs(board_repo) == before + + +def test_release_many_deletes_no_ref_when_one_lease_goes_stale_before_the_push( + tmp_path: Path, board_repo: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + board = _board(tmp_path, board_repo, "worker-a") + peer = _board(tmp_path, board_repo, "worker-b") + assert board.acquire_many(["kept", "stolen"], ttl=600) + push = board._cas_push_refs + observed: dict[str, dict[str, str]] = {} + + def peer_steals_first(updates: Sequence[tuple[str, str | None, str]], *, atomic: bool = False): + assert peer.acquire("stolen", ttl=600, steal=True) + observed["refs"] = _refs(board_repo) + return push(updates, atomic=atomic) + + monkeypatch.setattr(board, "_cas_push_refs", peer_steals_first) + result = board.release_many(["kept", "stolen"]) + + assert result == claims.ClaimBatchResult(False, ("stolen",), "changed concurrently") + assert _refs(board_repo) == observed["refs"] + assert board.holds("kept") + + +@pytest.mark.parametrize("method", ["acquire_many", "renew_many", "release_many"]) +@pytest.mark.parametrize( + ("keys", "error", "match"), + [ + (["first", "second", "first"], ValueError, "duplicate claim key 'first'"), + ([], ValueError, "at least one claim key is required"), + (["first", "../escape"], ValueError, "invalid claim key"), + ("first", TypeError, "not one string"), + ], +) +def test_batch_methods_reject_bad_keys_before_touching_the_board( + tmp_path: Path, board_repo: Path, method: str, keys: object, error: type[Exception], match: str +) -> None: + board = _board(tmp_path, board_repo, "worker-a") + + with pytest.raises(error, match=match): + getattr(board, method)(keys) + + assert not board.scratch.exists() + assert _refs(board_repo) == {} + + +def test_batch_refuses_a_board_without_atomic_push_support(tmp_path: Path, board_repo: Path) -> None: + with (board_repo / "config").open("a", encoding="utf-8") as config: + config.write("[receive]\n\tadvertiseAtomic = false\n") + board = _board(tmp_path, board_repo, "worker-a") + + with pytest.raises(claims.ClaimTransportError, match="does not support atomic pushes"): + board.acquire_many(["first", "second"], ttl=600) + + assert _refs(board_repo) == {} + assert board.acquire("first", ttl=600) + + +def test_single_key_push_is_unchanged_and_batches_push_atomically( + tmp_path: Path, board_repo: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + board = _board(tmp_path, board_repo, "worker-a") + git = board._git + pushes: list[list[str]] = [] + + def recording_git(args: list[str], **kwargs: object) -> subprocess.CompletedProcess[str]: + if args[0] == "push": + pushes.append(args) + return git(args, **kwargs) + + monkeypatch.setattr(board, "_git", recording_git) + assert board.acquire("single", ttl=600) + assert board.acquire_many(["one", "two"], ttl=600) + + single, one, two = (claims.CLAIM_REF_PREFIX + key for key in ("single", "one", "two")) + refs = _refs(board_repo) + assert pushes == [ + ["push", "--quiet", "--porcelain", f"--force-with-lease={single}:", board.repo_url, f"{refs[single]}:{single}"], + [ + "push", + "--quiet", + "--porcelain", + "--atomic", + f"--force-with-lease={one}:", + f"--force-with-lease={two}:", + board.repo_url, + f"{refs[one]}:{one}", + f"{refs[two]}:{two}", + ], + ] From de4dc3ef62afa8cb653b99ddb1a8cdcbaba4b93a Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 06:00:51 -0400 Subject: [PATCH 05/38] Name malformed leases as blocking keys in claim batches The B1 contract separates transport problems, which raise ClaimTransportError, from a lost race or a held or unverifiable claim, which return a failure that names the blocking keys. The batch methods raised MalformedLeaseError for an unverifiable lease, so the CLI printed a bare key-level error instead of the "no claim was ed" line. acquire_many, renew_many and release_many now report such a key as blocking with the reason "malformed lease". When keys block for different reasons, every key is named in batch order and the reasons are joined with "or". The per-key checks and the single-key methods are unchanged, and nothing is pushed. Tests cover each batch method with a malformed key, a batch with mixed reasons, the multi-node CLI line for a malformed claim, and the renew failure line. --- autoform_cli/claims.py | 54 ++++++++++++++++++++++++++++++----------- tests/test_claim_cli.py | 13 ++++++++++ tests/test_claims.py | 28 +++++++++++++++++++++ 3 files changed, 81 insertions(+), 14 deletions(-) diff --git a/autoform_cli/claims.py b/autoform_cli/claims.py index d6c44365..a76fa375 100644 --- a/autoform_cli/claims.py +++ b/autoform_cli/claims.py @@ -56,8 +56,9 @@ class MalformedLeaseError(ClaimTransportError): class ClaimBatchResult: """The outcome of an all-or-nothing multi-key claim operation. - It is true exactly when every key changed. Otherwise no key changed, - ``reason`` says why, and ``blocking`` names the keys responsible when known. + It is true exactly when the operation succeeded for every key. Otherwise no + key changed, ``reason`` says why, and ``blocking`` names the keys + responsible when known. """ ok: bool @@ -103,6 +104,11 @@ def _rejected_keys(detail: str, keys_by_ref: Mapping[str, str]) -> tuple[str, .. return tuple(key for ref, key in keys_by_ref.items() if ref in refs) +def _refusal(blocked: Mapping[str, str]) -> ClaimBatchResult: + """Refuse a batch before pushing, naming each blocking key in batch order and every reason.""" + return ClaimBatchResult(False, tuple(blocked), " or ".join(dict.fromkeys(blocked.values()))) + + def _is_finite_number(value: object) -> bool: if isinstance(value, bool): return False @@ -380,7 +386,9 @@ def release(self, key: str) -> bool: # The *_many methods read every key with one ls-remote, apply the per-key # checks of the single-key method, and change all keys in one atomic push, - # so a batch never leaves some of its claims taken and others not. + # so a batch never leaves some of its claims taken and others not. Where + # the single-key method raises MalformedLeaseError, a batch names that key + # as blocking instead, so one refusal can explain every blocked key. def acquire_many( self, @@ -394,9 +402,15 @@ def acquire_many( _validate_ttl(ttl) self._ensure_scratch() olds = self._remote_oids(keys) - held = tuple(key for key in keys if self._held_by_peer(key, olds.get(key))) - if held: - return ClaimBatchResult(False, held, "held by another worker") + blocked: dict[str, str] = {} + for key in keys: + try: + if self._held_by_peer(key, olds.get(key)): + blocked[key] = "held by another worker" + except MalformedLeaseError: + blocked[key] = "malformed lease" + if blocked: + return _refusal(blocked) updates = [(key, olds.get(key), self._make_lease_commit(key, ttl, note)) for key in keys] return self._cas_push_refs(updates, atomic=True) @@ -407,13 +421,19 @@ def renew_many(self, keys: Sequence[str], *, ttl: int | float = CLAIM_TTL_S) -> self._ensure_scratch() olds = self._remote_oids(keys) notes: dict[str, str] = {} + blocked: dict[str, str] = {} for key in keys: - lease = self._owned_lease(key, olds.get(key)) - if lease is not None: + try: + lease = self._owned_lease(key, olds.get(key)) + except MalformedLeaseError: + blocked[key] = "malformed lease" + continue + if lease is None: + blocked[key] = "not held by this worker" + else: notes[key] = str(lease.get("note", "")) - lost = tuple(key for key in keys if key not in notes) - if lost: - return ClaimBatchResult(False, lost, "not held by this worker") + if blocked: + return _refusal(blocked) updates = [(key, olds[key], self._make_lease_commit(key, ttl, notes[key])) for key in keys] return self._cas_push_refs(updates, atomic=True) @@ -423,9 +443,15 @@ def release_many(self, keys: Sequence[str]) -> ClaimBatchResult: self._ensure_scratch() olds = self._remote_oids(keys) present = [key for key in keys if key in olds] - foreign = tuple(key for key in present if self._owned_lease(key, olds[key]) is None) - if foreign: - return ClaimBatchResult(False, foreign, "not held by this worker") + blocked: dict[str, str] = {} + for key in present: + try: + if self._owned_lease(key, olds[key]) is None: + blocked[key] = "not held by this worker" + except MalformedLeaseError: + blocked[key] = "malformed lease" + if blocked: + return _refusal(blocked) if not present: return ClaimBatchResult(True) return self._cas_push_refs([(key, olds[key], "") for key in present], atomic=True) diff --git a/tests/test_claim_cli.py b/tests/test_claim_cli.py index d108066d..a09d4803 100644 --- a/tests/test_claim_cli.py +++ b/tests/test_claim_cli.py @@ -174,6 +174,10 @@ def test_claim_cli_multi_node_failure_changes_no_claim(tmp_path: Path, capsys) - assert main(_args(repo, scratch, "acquire", "a")) == 0 capsys.readouterr() + assert main(_args(repo, scratch, "renew", "a", "b")) == 1 + assert capsys.readouterr().out == ( + "error: could not renew a, b; no claim was renewed: not held by this worker: b\n" + ) assert main(_args(repo, scratch, "release", "a", "b")) == 1 assert capsys.readouterr().out == ( "error: could not release a, b; no claim was released: not held by this worker: b\n" @@ -181,6 +185,15 @@ def test_claim_cli_multi_node_failure_changes_no_claim(tmp_path: Path, capsys) - assert _claim_refs(repo) == sorted(CLAIM_REF_PREFIX + author_claim_key(node) for node in ("a", "b")) +def test_claim_cli_multi_node_names_a_malformed_claim(tmp_path: Path, capsys) -> None: + repo = _bare_repo(tmp_path) + _plant_message(repo, author_claim_key("b"), "not json") + + assert main(_args(repo, tmp_path / "scratch", "acquire", "a", "b")) == 1 + assert capsys.readouterr().out == "error: could not acquire a, b; no claim was acquired: malformed lease: b\n" + assert _claim_refs(repo) == [CLAIM_REF_PREFIX + author_claim_key("b")] + + def test_claim_cli_rejects_duplicate_targets_before_touching_the_board(tmp_path: Path, capsys) -> None: repo = _bare_repo(tmp_path) scratch = tmp_path / "scratch" diff --git a/tests/test_claims.py b/tests/test_claims.py index 99fd5fdf..84b3c939 100644 --- a/tests/test_claims.py +++ b/tests/test_claims.py @@ -653,6 +653,34 @@ def peer_steals_first(updates: Sequence[tuple[str, str | None, str]], *, atomic: assert board.holds("kept") +@pytest.mark.parametrize("method", ["acquire_many", "renew_many", "release_many"]) +def test_batch_names_a_malformed_lease_as_blocking_and_changes_nothing( + tmp_path: Path, board_repo: Path, method: str +) -> None: + board = _board(tmp_path, board_repo, "worker-a") + assert board.acquire("mine", ttl=600) + _plant_message(board_repo, "broken", "not json") + before = _refs(board_repo) + + result = getattr(board, method)(["mine", "broken"]) + + assert result == claims.ClaimBatchResult(False, ("broken",), "malformed lease") + assert _refs(board_repo) == before + + +def test_acquire_many_names_every_blocking_key_in_batch_order(tmp_path: Path, board_repo: Path) -> None: + peer = _board(tmp_path, board_repo, "worker-b") + assert peer.acquire("held", ttl=600) + _plant_message(board_repo, "broken", "not json") + before = _refs(board_repo) + board = _board(tmp_path, board_repo, "worker-a") + + result = board.acquire_many(["broken", "free", "held"], ttl=600) + + assert result == claims.ClaimBatchResult(False, ("broken", "held"), "malformed lease or held by another worker") + assert _refs(board_repo) == before + + @pytest.mark.parametrize("method", ["acquire_many", "renew_many", "release_many"]) @pytest.mark.parametrize( ("keys", "error", "match"), From acf9f19124658e1374a5856384b87c11242cdf40 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 05:54:34 -0400 Subject: [PATCH 06/38] Report articles whose Lean target is deprecated The Lean findings pass of audit gains lean-target-deprecated: an article's lean: declaration whose source carries the deprecated attribute in an @[...] list before its keyword, on the declaration line or on the contiguous attribute lines directly above it. Comments are blanked first and string literals inside attribute lists are ignored, so a commented-out attribute or a name such as deprecated_alias does not count. --- autoform_cli/audit.py | 59 ++++++++++++++++++++++++++++++++++++++++++- tests/test_audit.py | 52 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 110 insertions(+), 1 deletion(-) diff --git a/autoform_cli/audit.py b/autoform_cli/audit.py index 35e427cc..afd6ef42 100644 --- a/autoform_cli/audit.py +++ b/autoform_cli/audit.py @@ -8,6 +8,7 @@ from __future__ import annotations import json +import re import statistics from bisect import bisect_right from dataclasses import asdict, dataclass @@ -16,7 +17,7 @@ from . import status from .coverage import CoverageSummary, load_coverage from .graph import Graph, GraphValidationError, Node, load_graph -from .lean import SourceIndex, declaration_names, index_project +from .lean import _DECLARATION, Declaration, SourceIndex, _without_lean_comments, declaration_names, index_project from .markdown import FENCE as _FENCE from .markdown import frontmatter_end as _frontmatter_end from .markdown import HEADING as _HEADING @@ -51,6 +52,17 @@ "theorem": frozenset({"lemma", "theorem"}), } +#: One ``@[...]`` attribute list; a string literal inside it may hold brackets. +_ATTRIBUTE_LIST = r'@\[(?:[^\]"]|"(?:[^"\\]|\\.)*")*\]' +#: The attribute lists and modifiers a declaration keyword closes, starting at +#: a line start, so contiguous attribute lines above the keyword's line count. +_DECLARATION_HEADER = re.compile( + rf"(?:\A|\n)[ \t]*((?:{_ATTRIBUTE_LIST}\s*)*)" + r"(?:(?:private|protected|noncomputable|partial|unsafe|scoped|local)\s+)*\Z" +) +_STRING_LITERAL = re.compile(r'"(?:[^"\\]|\\.)*"') +_DEPRECATED_ATTRIBUTE = re.compile(r"(?:\[|,)\s*(?:(?:scoped|local)\s+)?deprecated\b") + @dataclass(frozen=True, order=True, slots=True) class AuditFinding: @@ -366,6 +378,7 @@ def _lean_findings(graph: Graph, lean_root: str | Path) -> list[AuditFinding]: index = index_project(root) spans = _source_spans(index) sizes: dict[str, int] = {} + sources: dict[Path, list[str] | None] = {} for node_id in sorted(graph.nodes): node = graph.nodes[node_id] article_path = _relative_path(node.path, graph.blueprint_dir) @@ -394,6 +407,16 @@ def _lean_findings(graph: Graph, lean_root: str | Path) -> list[AuditFinding]: else: resolved.append(declaration) + for declaration in resolved: + if _declared_deprecated(declaration, index.root, sources): + findings.append( + AuditFinding( + article_path, + "lean-target-deprecated", + f"lean target {declaration.name} is deprecated; point lean: at its replacement", + ) + ) + expected = _DECLARATION_KEYWORDS.get((node.declaration or "").casefold()) if expected and resolved and not any(declaration.keyword in expected for declaration in resolved): actual = ", ".join(sorted({declaration.keyword for declaration in resolved})) @@ -412,6 +435,40 @@ def _lean_findings(graph: Graph, lean_root: str | Path) -> list[AuditFinding]: return findings +def _declared_deprecated(declaration: Declaration, root: Path, sources: dict[Path, list[str] | None]) -> bool: + """Whether the source declaration carries the ``deprecated`` attribute lexically. + + Only ``@[...]`` lists before the keyword count: on the declaration's line, + or on the contiguous attribute lines directly above it. Comments are + blanked first, so a commented-out attribute does not count. + """ + + if declaration.path not in sources: + try: + text = (root / declaration.path).read_text(encoding="utf-8") + except (OSError, UnicodeError): + sources[declaration.path] = None + else: + sources[declaration.path] = _without_lean_comments(text).splitlines() + lines = sources[declaration.path] + if lines is None or not 0 < declaration.line <= len(lines): + return False + line = lines[declaration.line - 1] + keyword = _DECLARATION.match(line) + if keyword is None: + return False + # The attribute lines cannot reach above the previous declaration, so the + # search never rescans the whole file. + first = declaration.line - 1 + while first and not _DECLARATION.match(lines[first - 1]): + first -= 1 + header = _DECLARATION_HEADER.search("\n".join([*lines[first : declaration.line - 1], line[: keyword.start(1)]])) + return header is not None and any( + _DEPRECATED_ATTRIBUTE.search(_STRING_LITERAL.sub('""', attributes)) + for attributes in re.findall(_ATTRIBUTE_LIST, header.group(1)) + ) + + def _source_spans(index: SourceIndex) -> dict[str, int]: """Measure each declaration's source span, up to the next declaration. diff --git a/tests/test_audit.py b/tests/test_audit.py index d2bc1091..197f12bd 100644 --- a/tests/test_audit.py +++ b/tests/test_audit.py @@ -254,6 +254,58 @@ def test_audit_validates_lean_targets_only_when_root_is_supplied(tmp_path: Path) ] +def test_audit_reports_lean_targets_declared_deprecated(tmp_path: Path) -> None: + blueprint = tmp_path / "blueprint" + _coverage(blueprint) + targets = { + "same-line.md": "Project.sameLine", + "line-above.md": "Project.lineAbove", + "multi-line.md": "Project.multiLine", + "fresh.md": "Project.fresh", + "decoy.md": "Project.decoy", + } + for relative, name in targets.items(): + _article(blueprint, relative, declaration="theorem", statement="formalized", lean=name) + lean_root = tmp_path / "lean" + lean_root.mkdir() + source = [ + "theorem Project.fresh : True := trivial", + "", + '@[deprecated Project.fresh (since := "2025-01-01")] theorem Project.sameLine : True := trivial', + "", + "/-- Kept for one release. -/", + "@[simp]", + '@[deprecated Project.fresh (since := "2025-01-01")]', + "protected theorem Project.lineAbove : True := trivial", + "", + "@[simp,", + ' deprecated "use [Project.fresh] instead" (since := "2025-01-01")]', + "-- removed after the next release", + "theorem Project.multiLine : True := trivial", + "", + '-- @[deprecated Project.fresh (since := "2025-01-01")]', + "/- @[deprecated Project.fresh] -/", + '@[to_additive "deprecated", deprecated_alias]', + "theorem Project.decoy : True := by", + ' have : "@[deprecated]" = "@[deprecated]" := rfl', + " trivial", + ] + (lean_root / "Old.lean").write_text("\n".join(source) + "\n", encoding="utf-8") + + without_lean = _finding_map(blueprint) + with_lean = _finding_map(blueprint, lean_root=lean_root) + + def deprecated(name: str) -> list[tuple[str, str]]: + return [("lean-target-deprecated", f"lean target {name} is deprecated; point lean: at its replacement")] + + assert without_lean == {} + assert with_lean == { + "roadmap/same-line.md": deprecated("Project.sameLine"), + "roadmap/line-above.md": deprecated("Project.lineAbove"), + "roadmap/multi-line.md": deprecated("Project.multiLine"), + } + + def test_audit_reports_invalid_lean_root_once(tmp_path: Path) -> None: blueprint = tmp_path / "blueprint" _coverage(blueprint) From c644d70de9dff8145d331d287e5f0abe3147606e Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 05:55:07 -0400 Subject: [PATCH 07/38] Add work impact to show what revising shared declarations affects autoform work impact SELECTOR --lean-root PATH reads every project-local constant from the built environment with a new Lean probe, run through skeleton.run_probe so the freshness check, time limit and output bound apply. The revised set is the article's lean: names or the repeatable --declaration names, each of which must be project-local. Statement impact is the reverse closure over meaning edges (types, the values of definitions and opaque constants, an inductive's constructors). Proof impact covers constants with a body whose value reaches that set directly or through internal details such as simp's _simp_1 companions and a definition's _proof_1. The report names the impacted articles, the helpers no article names with their location and owning article, impacted articles with no Markdown dependency path to the revised one, deprecated constants and their users, the claim targets, and whether the revision is contained. --json writes the autoform-impact/v1 schema. The probe imports the modules Lake builds: each library's globs (M, M.* and M.+; other forms fail closed), else its roots. LeanLibrary records the globs, and run_probe takes a label so its errors name the impact probe; skeleton behavior is unchanged. --- autoform_cli/__main__.py | 61 ++ autoform_cli/impact.py | 702 +++++++++++++++++++ autoform_cli/probes/impact_probe.lean | 111 +++ autoform_cli/skeleton.py | 18 +- tests/test_impact.py | 955 ++++++++++++++++++++++++++ 5 files changed, 1843 insertions(+), 4 deletions(-) create mode 100644 autoform_cli/impact.py create mode 100644 autoform_cli/probes/impact_probe.lean create mode 100644 tests/test_impact.py diff --git a/autoform_cli/__main__.py b/autoform_cli/__main__.py index 895c5fe4..2bbcf8ef 100644 --- a/autoform_cli/__main__.py +++ b/autoform_cli/__main__.py @@ -20,6 +20,7 @@ from .doctor import diagnose_project from .dashboard import publication_bound_live_state, serve_dashboard from .graph import GraphValidationError, load_graph +from .impact import ImpactError, format_impact, revision_impact from .lean import build_linker, declaration_names from .project import ProjectCatalogError, inspect_project, load_release_catalog from .render import PublicationError, render_site @@ -139,6 +140,30 @@ def main(argv: Sequence[str] | None = None) -> int: "target", nargs="?", default=".", help="project root or blueprint directory" ) work_assumptions.add_argument("--json", action="store_true", help="write stable machine-readable output") + work_impact = work_subparsers.add_parser( + "impact", help="show which articles and helpers a revision of an article's Lean declarations affects" + ) + work_impact.add_argument("selector", help="path-derived node id or durable article_id") + work_impact.add_argument("target", nargs="?", default=".", help="project root or blueprint directory") + work_impact.add_argument( + "--lean-root", type=Path, required=True, metavar="PATH", help="the built Lean project" + ) + work_impact.add_argument( + "--declaration", + action="append", + default=[], + dest="declarations", + metavar="NAME", + help="revise this project-local constant instead of the article's lean: declarations (repeatable)", + ) + work_impact.add_argument("--json", action="store_true", help="write stable machine-readable output") + work_impact.add_argument( + "--timeout", + type=_positive_seconds, + metavar="SECONDS", + help=f"seconds the Lean probe may run (default {DEFAULT_PROBE_TIMEOUT:g}); " + "the Lake freshness check before it has its own budget", + ) claim = subparsers.add_parser("claim", help="coordinate temporary node ownership through Git refs") claim_subparsers = claim.add_subparsers(dest="claim_command", required=True) for operation in ("acquire", "renew", "release"): @@ -438,6 +463,8 @@ def _project(args: argparse.Namespace) -> int: def _work(args: argparse.Namespace) -> int: if args.work_command == "assumptions": return _work_assumptions(args) + if args.work_command == "impact": + return _work_impact(args) # Only loading the roadmap can fail on the project's paths; printing the # result stays outside, so an output error is not reported as one. try: @@ -542,6 +569,40 @@ def _work_assumptions(args: argparse.Namespace) -> int: return 0 +def _work_impact(args: argparse.Namespace) -> int: + try: + report = revision_impact( + args.target, + args.selector, + lean_root=args.lean_root, + declarations=args.declarations, + timeout=args.timeout, + ) + except (GraphValidationError, RuntimeProjectionError) as error: + for issue in error.issues: + print(f"error: {_human_text(issue)}", file=sys.stderr) + return 2 + except (WorkError, ImpactError) as error: + print(f"error: {_human_text(error)}", file=sys.stderr) + return 2 + except SkeletonError as error: + # Probe failures carry Lean's multi-line output; escape it line by line. + for issue in error.issues: + for index, line in enumerate(str(issue).splitlines() or [""]): + print(("error: " if index == 0 else "") + _human_text(line), file=sys.stderr) + return 2 + except (OSError, RuntimeError, ValueError): + print("error: project, blueprint, or Lean root path cannot be read", file=sys.stderr) + return 2 + + if args.json: + print(report.to_json()) + return 0 + for line in format_impact(report): + print(_human_text(line)) + return 0 + + def _print_project_inspection(result) -> None: if result.project_root is not None: print(f"Project root: {_human_text(result.project_root)}") diff --git a/autoform_cli/impact.py b/autoform_cli/impact.py new file mode 100644 index 00000000..6c2dd1e1 --- /dev/null +++ b/autoform_cli/impact.py @@ -0,0 +1,702 @@ +"""Report what revising an article's Lean declarations would affect. + +``autoform work impact`` answers the question a worker must settle before +changing declarations other work may build on: which articles now state +something else, whose proofs may stop elaborating, which helpers no article +names sit in between, and whether the Markdown roadmap records those +dependencies. The answer is read from the elaborated environment by a Lean +probe rather than from source text, so uses that only automation such as +``simp`` introduces are seen too. +""" + +from __future__ import annotations + +import json +import os +import re +from collections.abc import Callable, Iterable, Mapping, Sequence +from dataclasses import dataclass +from pathlib import Path + +from . import skeleton +from .lean import SourceIndex, index_project +from .runtime import load_runtime_graph +from .skeleton import LeanLibrary, SkeletonError, _lean_name, _lean_name_parts, lean_libraries, path_of +from .work import work_context + +IMPACT_SCHEMA = "autoform-impact/v1" + +#: Every line the impact probe wants read back starts with this marker. +IMPACT_MARKER = "AUTOFORM_IMPACT " + +_KINDS = frozenset({"axiom", "constructor", "def", "inductive", "opaque", "quot", "recursor", "theorem"}) +#: Kinds whose value belongs to their meaning, as in the skeleton probe's +#: ``meaningConstants``; an inductive's constructors stand in for its value. +_VALUE_IS_MEANING = frozenset({"def", "inductive", "opaque"}) +#: Kinds with a body that elaborates against the statements it uses. +_HAS_BODY = frozenset({"def", "opaque", "theorem"}) + + +class ImpactError(ValueError): + """The question cannot be asked: nothing to revise, or a name that is not local.""" + + +# --------------------------------------------------------------------------- # +# Probe records +# --------------------------------------------------------------------------- # + + +@dataclass(frozen=True, slots=True) +class ConstantRecord: + """One project-local constant as the impact probe read it. + + ``type_uses`` and ``value_uses`` name only project-local constants. + ``internal`` is ``Name.isInternalDetail`` of the user-facing name, so a + private declaration someone wrote is not internal, while the companions + Lean generates (``_proof_1``, ``match_1``, ``_simp_1``) are. + """ + + name: str + kind: str + module: str + type_uses: tuple[str, ...] = () + value_uses: tuple[str, ...] = () + instance: bool = False + internal: bool = False + parent: str | None = None + deprecated: bool = False + replacement: str | None = None + uses_deprecated: tuple[str, ...] = () + value_missing: bool = False + + @property + def meaning_uses(self) -> tuple[str, ...]: + """The local constants this constant's meaning rests on.""" + + return self.type_uses + self.value_uses if self.kind in _VALUE_IS_MEANING else self.type_uses + + +_RECORD_FIELDS: dict[str, tuple[type, ...]] = { + "deprecated": (bool,), + "instance": (bool,), + "internal": (bool,), + "kind": (str,), + "module": (str,), + "name": (str,), + "parent": (str, type(None)), + "replacement": (str, type(None)), + "type_uses": (list,), + "uses_deprecated": (list,), + "value_missing": (bool,), + "value_uses": (list,), +} + + +def parse_impact_output(text: str) -> dict[str, ConstantRecord]: + """Return the probe's strictly validated records, keyed by constant name. + + A record whose value the probe could not read fails closed: the users of + whatever that value mentions could not be traced. + """ + + records: dict[str, ConstantRecord] = {} + for line in text.splitlines(): + if not line.startswith(IMPACT_MARKER): + continue + try: + payload = json.loads(line[len(IMPACT_MARKER) :]) + except json.JSONDecodeError as exc: + raise SkeletonError([f"the impact probe emitted invalid JSON: {exc}"]) from exc + record = _constant_record(payload) + if record.name in records: + raise SkeletonError([f"the impact probe emitted {record.name} twice"]) + records[record.name] = record + if not records: + raise SkeletonError( + ["the impact probe found no project-local constants; are the library roots and globs built?"] + ) + for record in records.values(): + if record.value_missing: + raise SkeletonError([f"the impact probe could not read the value of {record.name}"]) + unknown = sorted(set(record.type_uses + record.value_uses) - records.keys()) + if unknown: + raise SkeletonError([f"the impact probe reported {record.name} using unknown constants: {unknown}"]) + return records + + +def _constant_record(payload: object) -> ConstantRecord: + if not isinstance(payload, dict) or set(payload) != set(_RECORD_FIELDS): + raise SkeletonError(["the impact probe emitted a record with unexpected fields"]) + for field, expected in _RECORD_FIELDS.items(): + value = payload[field] + if not isinstance(value, expected) or ( + isinstance(value, list) and not all(isinstance(item, str) for item in value) + ): + raise SkeletonError([f"the impact probe emitted a malformed {field!r} field"]) + if not payload["name"] or not payload["module"] or payload["kind"] not in _KINDS: + raise SkeletonError([f"the impact probe emitted a malformed record for {payload['name']!r}"]) + return ConstantRecord( + name=payload["name"], + kind=payload["kind"], + module=payload["module"], + type_uses=tuple(payload["type_uses"]), + value_uses=tuple(payload["value_uses"]), + instance=payload["instance"], + internal=payload["internal"], + parent=payload["parent"], + deprecated=payload["deprecated"], + replacement=payload["replacement"], + uses_deprecated=tuple(payload["uses_deprecated"]), + value_missing=payload["value_missing"], + ) + + +def _name_key(name: str) -> tuple[object, ...]: + """Compare names by component, so ``A.«b»`` and ``A.b`` are the same name.""" + + try: + parts = _lean_name_parts(name) + except SkeletonError: + return (name,) + return tuple((part, not quoted and part.isascii() and part.isdigit()) for part, quoted in parts) + + +# --------------------------------------------------------------------------- # +# Project modules and the probe +# --------------------------------------------------------------------------- # + +#: Module name components the probe can import without quoting. +_MODULE_COMPONENT = re.compile(r"[A-Za-z_][A-Za-z0-9_'!?]*") + + +def project_modules(libraries: Sequence[LeanLibrary]) -> tuple[tuple[str, ...], tuple[str, ...]]: + """Return the modules Lake builds for ``libraries`` and the local module prefixes. + + A library's ``globs`` select its modules as Lake's ``Glob`` does: ``M`` is + one module, ``M.*`` the module and its submodules, ``M.+`` its strict + submodules, found by walking the source tree like ``forEachModuleInDir``. + A library without globs builds its roots. A module is local when it equals + or lies under a library root or a glob base. + """ + + modules: set[str] = set() + prefixes: set[str] = set() + for library in libraries: + prefixes.update(library.roots) + for glob in library.globs or library.roots: + base, mode = _glob(glob, library) + prefixes.add(base) + directory = library.src_dir.joinpath(*base.split(".")) + if mode != "+": + if not directory.with_suffix(".lean").is_file(): + raise SkeletonError([f"lean_lib {library.name}: module {base} has no source file"]) + modules.add(base) + if mode: + if not directory.is_dir(): + raise SkeletonError( + [f"lean_lib {library.name}: glob {glob} names no source directory {base.replace('.', '/')}"] + ) + modules.update(f"{base}.{module}" for module in _submodules(directory, library)) + if not modules: + raise SkeletonError(["the Lake configuration selects no modules"]) + return tuple(sorted(modules)), tuple(sorted(prefixes)) + + +def _glob(glob: str, library: LeanLibrary) -> tuple[str, str]: + base, mode = glob, "" + if glob.endswith((".*", ".+")): + base, mode = glob[:-2], glob[-1] + if not base or not all(_MODULE_COMPONENT.fullmatch(part) for part in base.split(".")): + raise SkeletonError( + [ + f"lean_lib {library.name}: cannot read module glob {glob!r}; the impact probe supports " + "`M`, `M.*` and `M.+` over plain module names" + ] + ) + return base, mode + + +def _submodules(directory: Path, library: LeanLibrary) -> list[str]: + """Every ``.lean`` file below ``directory`` as a relative module name.""" + + found: list[str] = [] + visited: set[Path] = set() + + def walk(path: Path, prefix: tuple[str, ...]) -> None: + resolved = path.resolve() + if resolved in visited: + return + visited.add(resolved) + with os.scandir(path) as entries: + for entry in sorted(entries, key=lambda item: item.name): + if entry.is_dir(): + walk(Path(entry.path), (*prefix, entry.name)) + elif Path(entry.name).suffix == ".lean": + parts = (*prefix, Path(entry.name).stem) + if not all(_MODULE_COMPONENT.fullmatch(part) for part in parts): + raise SkeletonError( + [ + f"lean_lib {library.name}: cannot import {entry.path}; the impact probe supports " + "plain module names only" + ] + ) + found.append(".".join(parts)) + + walk(directory, ()) + return found + + +def _impact_template() -> str: + """The Lean impact probe source, kept under probes/ beside this module.""" + + return (Path(__file__).parent / "probes" / "impact_probe.lean").read_text(encoding="utf-8") + + +def render_impact_probe(*, imports: Iterable[str], project_roots: Iterable[str]) -> str: + """Render the Lean program that records every project-local constant.""" + + modules = sorted(set(imports)) + if not modules: + raise SkeletonError(["refusing to render an impact probe with no imports"]) + return _impact_template().format( + imports="\n".join(f"import {module}" for module in modules), + marker=IMPACT_MARKER, + output_env=skeleton.PROBE_OUTPUT_ENV, + output_limit=skeleton.DEFAULT_PROBE_OUTPUT_LIMIT, + project_roots=", ".join(_lean_name(name) for name in sorted(set(project_roots))), + ) + + +# --------------------------------------------------------------------------- # +# The impact computation +# --------------------------------------------------------------------------- # + + +@dataclass(frozen=True, slots=True) +class ImpactArticle: + """An article as the impact computation sees it: its Lean names and Markdown dependencies.""" + + id: str + article_id: str | None + declarations: tuple[str, ...] + dependencies: tuple[str, ...] = () + + @property + def claim_target(self) -> str: + return self.article_id or self.id + + +@dataclass(frozen=True, slots=True) +class ImpactedArticle: + """An article other than the revised one, with its impacted declarations.""" + + id: str + article_id: str | None + claim_target: str + declarations: tuple[str, ...] + + def as_dict(self) -> dict[str, object]: + return { + "id": self.id, + "article_id": self.article_id, + "claim_target": self.claim_target, + "declarations": list(self.declarations), + } + + +@dataclass(frozen=True, slots=True) +class ImpactHelper: + """An impacted constant that no article names and that is not an internal detail.""" + + name: str + kind: str + impact: str + module: str + path: str | None + line: int | None + owner: str | None + + def as_dict(self) -> dict[str, object]: + return { + "name": self.name, + "kind": self.kind, + "impact": self.impact, + "module": self.module, + "path": self.path, + "line": self.line, + "owner": self.owner, + } + + +@dataclass(frozen=True, slots=True) +class DeprecatedConstant: + """A local deprecated constant, its replacement, and the local constants that use it.""" + + name: str + replacement: str | None + users: tuple[str, ...] + + def as_dict(self) -> dict[str, object]: + return {"name": self.name, "replacement": self.replacement, "users": list(self.users)} + + +@dataclass(frozen=True, slots=True) +class ImpactReport: + """What revising ``declarations`` of ``article`` affects.""" + + source_revision: str + article: ImpactArticle + declarations: tuple[str, ...] + statement_impacted: tuple[ImpactedArticle, ...] + proof_impacted: tuple[ImpactedArticle, ...] + helpers: tuple[ImpactHelper, ...] + undeclared_dependencies: tuple[str, ...] + deprecated: tuple[DeprecatedConstant, ...] + deprecated_unused: tuple[str, ...] + claim_targets: tuple[str, ...] + + @property + def contained(self) -> bool: + """Whether nothing outside the revised article is impacted.""" + + return not (self.statement_impacted or self.proof_impacted or self.helpers) + + def as_dict(self) -> dict[str, object]: + return { + "schema": IMPACT_SCHEMA, + "source_revision": self.source_revision, + "article": { + "id": self.article.id, + "article_id": self.article.article_id, + "claim_target": self.article.claim_target, + }, + "declarations": list(self.declarations), + "contained": self.contained, + "statement_impacted": [item.as_dict() for item in self.statement_impacted], + "proof_impacted": [item.as_dict() for item in self.proof_impacted], + "helpers": [helper.as_dict() for helper in self.helpers], + "undeclared_dependencies": list(self.undeclared_dependencies), + "deprecated": [item.as_dict() for item in self.deprecated], + "deprecated_unused": list(self.deprecated_unused), + "claim_targets": list(self.claim_targets), + } + + def to_json(self) -> str: + return json.dumps(self.as_dict(), sort_keys=True, separators=(",", ":")) + + +Locator = Callable[[ConstantRecord], tuple[str | None, int | None]] + + +def compute_impact( + records: Mapping[str, ConstantRecord], + articles: Sequence[ImpactArticle], + revised: ImpactArticle, + declarations: Sequence[str], + *, + source_revision: str, + locate: Locator | None = None, +) -> ImpactReport: + """Compute what revising ``declarations`` of article ``revised`` affects. + + The statement-impacted set M is the reverse closure of the revised names + over meaning edges: a constant's type, plus the value of a definition or + opaque constant and the constructors of an inductive. The proof-impacted + set P holds the constants with a body, outside M, whose value uses M + directly or through internal-detail constants, such as the ``_simp_1`` + companion ``simp`` uses or the ``_proof_1`` a definition's nested proof + becomes. Definitions belong in P too: their nested proofs can stop + elaborating although their meaning is unchanged. + """ + + by_key = {_name_key(name): name for name in records} + revised_names: dict[str, str] = {} + missing: list[str] = [] + for name in declarations: + record_name = by_key.get(_name_key(name)) + if record_name is None: + missing.append(name) + else: + revised_names.setdefault(record_name, name) + if missing: + raise ImpactError("not a project-local constant: " + ", ".join(dict.fromkeys(missing))) + if not revised_names: + raise ImpactError(f"{revised.id}: nothing to revise; pass --declaration NAME") + + meaning_users: dict[str, set[str]] = {} + users: dict[str, set[str]] = {} + for record in records.values(): + for used in record.meaning_uses: + meaning_users.setdefault(used, set()).add(record.name) + for used in (*record.type_uses, *record.value_uses): + users.setdefault(used, set()).add(record.name) + + meaning = _reverse_closure(revised_names, meaning_users) + through_internal = _reverse_closure(meaning, users, admit=lambda name: records[name].internal) + proof = { + record.name + for record in records.values() + if record.kind in _HAS_BODY + and record.name not in meaning + and any(used in through_internal for used in record.value_uses) + } + + named: dict[str, list[str]] = {} + statement_impacted: list[ImpactedArticle] = [] + proof_impacted: list[ImpactedArticle] = [] + for article in articles: + resolved = [(name, by_key.get(_name_key(name))) for name in article.declarations] + for _, record_name in resolved: + if record_name is not None: + named.setdefault(record_name, []).append(article.id) + if article.id == revised.id: + continue + in_meaning = tuple(name for name, record_name in resolved if record_name in meaning) + in_proof = tuple(name for name, record_name in resolved if record_name in proof) + if in_meaning: + statement_impacted.append( + ImpactedArticle(article.id, article.article_id, article.claim_target, in_meaning) + ) + elif in_proof: + proof_impacted.append(ImpactedArticle(article.id, article.article_id, article.claim_target, in_proof)) + statement_impacted.sort(key=lambda item: item.id) + proof_impacted.sort(key=lambda item: item.id) + + helpers: list[ImpactHelper] = [] + for name in sorted(meaning | proof): + record = records[name] + if record.internal or name in named or name in revised_names: + continue + path, line = locate(record) if locate is not None else (None, None) + impact = "statement" if name in meaning else "proof" + helpers.append(ImpactHelper(name, record.kind, impact, record.module, path, line, _owner(record, records, named))) + + by_id = {article.id: article for article in articles} + impacted = (*statement_impacted, *proof_impacted) + undeclared = sorted(item.id for item in impacted if not _reaches(item.id, revised.id, by_id)) + + deprecated_users: dict[str, set[str]] = {} + for record in records.values(): + for used in record.uses_deprecated: + deprecated_users.setdefault(used, set()).add(record.name) + deprecated = tuple( + DeprecatedConstant( + record.name, + record.replacement, + _users_through_internal(record.name, deprecated_users.get(record.name, set()), users, records), + ) + for record in sorted(records.values(), key=lambda item: item.name) + if record.deprecated + ) + + claim_targets = (revised.claim_target, *sorted({item.claim_target for item in impacted} - {revised.claim_target})) + return ImpactReport( + source_revision=source_revision, + article=revised, + declarations=tuple(revised_names.values()), + statement_impacted=tuple(statement_impacted), + proof_impacted=tuple(proof_impacted), + helpers=tuple(helpers), + undeclared_dependencies=tuple(undeclared), + deprecated=deprecated, + deprecated_unused=tuple(item.name for item in deprecated if not item.users), + claim_targets=claim_targets, + ) + + +def _reverse_closure( + start: Iterable[str], + users: Mapping[str, set[str]], + *, + admit: Callable[[str], bool] = lambda name: True, +) -> set[str]: + """``start`` and everything that reaches it through admitted users.""" + + reached = set(start) + work = list(reached) + while work: + for user in users.get(work.pop(), ()): + if user not in reached and admit(user): + reached.add(user) + work.append(user) + return reached + + +def _owner(record: ConstantRecord, records: Mapping[str, ConstantRecord], named: Mapping[str, list[str]]) -> str | None: + """The article naming the nearest ``parent`` ancestor of ``record``, if any.""" + + seen: set[str] = set() + parent = record.parent + while parent is not None and parent not in seen: + if parent in named: + return min(named[parent]) + seen.add(parent) + ancestor = records.get(parent) + parent = ancestor.parent if ancestor is not None else None + return None + + +def _reaches(start: str, target: str, articles: Mapping[str, ImpactArticle]) -> bool: + """Whether ``start`` reaches ``target`` through Markdown dependencies.""" + + seen = {start} + work = [start] + while work: + article = articles.get(work.pop()) + for dependency in article.dependencies if article is not None else (): + if dependency == target: + return True + if dependency not in seen: + seen.add(dependency) + work.append(dependency) + return False + + +def _users_through_internal( + name: str, + direct: set[str], + users: Mapping[str, set[str]], + records: Mapping[str, ConstantRecord], +) -> tuple[str, ...]: + """The local users of ``name``; an internal-detail user stands for its own users. + + An internal-detail user nothing else uses is kept, so that a constant is + never reported unused while something still mentions it. + """ + + found: set[str] = set() + seen = {name} + work = sorted(direct) + while work: + user = work.pop() + if user in seen: + continue + seen.add(user) + further = users.get(user, set()) - {user} + if records[user].internal and further: + work.extend(sorted(further)) + else: + found.add(user) + return tuple(sorted(found)) + + +# --------------------------------------------------------------------------- # +# The query +# --------------------------------------------------------------------------- # + + +def revision_impact( + project_or_blueprint: str | Path, + selector: str, + *, + lean_root: str | Path, + declarations: Sequence[str] = (), + timeout: float | None = None, +) -> ImpactReport: + """Report what revising the Lean declarations of the selected article would affect. + + ``declarations`` replaces the article's ``lean:`` names as the revised set. + The Lean project must be built and fresh: the probe runs through + ``skeleton.run_probe``, with its freshness check, time limit and output bound. + """ + + source_revision, item = work_context(project_or_blueprint, selector) + runtime = load_runtime_graph(project_or_blueprint) + articles = { + node.id: ImpactArticle( + node.id, + node.article_id, + tuple(dict.fromkeys(target.declaration for target in node.lean_targets)), + tuple(node.dependencies), + ) + for node in runtime.nodes + } + revised = articles.get(item.node_id) + if runtime.source_revision != source_revision or revised is None: + raise ImpactError("the roadmap changed while it was read; rerun the command") + names = tuple(dict.fromkeys(declarations)) or revised.declarations + if not names: + raise ImpactError(f"{revised.id} names no lean: declaration; pass --declaration NAME") + + root = Path(lean_root).expanduser().resolve() + libraries = lean_libraries(root) + modules, prefixes = project_modules(libraries) + output = skeleton.run_probe( + render_impact_probe(imports=modules, project_roots=prefixes), + root, + timeout=skeleton.DEFAULT_PROBE_TIMEOUT if timeout is None else timeout, + label="impact probe", + ) + return compute_impact( + parse_impact_output(output), + tuple(articles.values()), + revised, + names, + source_revision=source_revision, + locate=_locator(libraries, root), + ) + + +def _locator(libraries: tuple[LeanLibrary, ...], root: Path) -> Locator: + """Locate a constant through the lexical source index, else by its module's file. + + The index is built on first use, since a contained revision locates nothing. + """ + + index: SourceIndex | None = None + + def locate(record: ConstantRecord) -> tuple[str | None, int | None]: + nonlocal index + if index is None: + index = index_project(root) + module_path = path_of(record.module, libraries, root) + declaration = index.find(record.name.removeprefix(f"_private.{record.module}.0.")) + if declaration is not None and module_path in (None, declaration.path.as_posix()): + return declaration.path.as_posix(), declaration.line + return module_path, None + + return locate + + +def format_impact(report: ImpactReport) -> list[str]: + """The report as short lines of text; callers escape them for the terminal.""" + + names = ", ".join(report.declarations) + lines = [ + f"Revising {names} of {_label(report.article.id, report.article.article_id)}", + f"Graph source revision: {report.source_revision}", + ] + if report.contained: + lines.append(f"Contained: no other article or helper uses {names}, so it can be revised in place.") + for title, impacted in ( + ("Statement impacted", report.statement_impacted), + ("Proof impacted", report.proof_impacted), + ): + if impacted: + lines.append(f"{title}:") + lines.extend(f" {_label(item.id, item.article_id)}: {', '.join(item.declarations)}" for item in impacted) + if report.helpers: + lines.append("Helpers no article names:") + for helper in report.helpers: + where = helper.path or helper.module + if helper.path and helper.line is not None: + where = f"{where}:{helper.line}" + owner = f"owner {helper.owner}" if helper.owner else "no owner" + lines.append(f" {helper.name} ({helper.kind}, {helper.impact}) {where}; {owner}") + if report.undeclared_dependencies: + lines.append( + "Impacted without a Markdown dependency path to the revised article: " + + ", ".join(report.undeclared_dependencies) + ) + if report.deprecated: + lines.append("Deprecated:") + for item in report.deprecated: + replacement = f" -> {item.replacement}" if item.replacement else "" + users = f"used by {', '.join(item.users)}" if item.users else "no users, safe to delete" + lines.append(f" {item.name}{replacement}: {users}") + lines.append("Claim targets: " + ", ".join(report.claim_targets)) + return lines + + +def _label(node_id: str, article_id: str | None) -> str: + return f"{node_id} [{article_id}]" if article_id else node_id diff --git a/autoform_cli/probes/impact_probe.lean b/autoform_cli/probes/impact_probe.lean new file mode 100644 index 00000000..c86f9da3 --- /dev/null +++ b/autoform_cli/probes/impact_probe.lean @@ -0,0 +1,111 @@ +{imports} +-- Autoform impact probe. This file is a Python-format template: `{{`/`}}` are +-- literal braces and single-brace fields are filled by autoform_cli.impact. +-- It is written to a temporary file and run with `lake env lean` inside the +-- built project; it never modifies the project. +import Lean.Linter.Deprecated +import Lean.Elab.Command +import Lean.Data.Json + +open Lean Elab Command Meta + +namespace AutoformImpact + +def probeOutputLimit : Nat := {output_limit} + +/-- Write one complete record without letting the scratch file grow past the +CLI's output limit. The Python reader checks the limit again after exit. -/ +def emitRecord (out : IO.FS.Handle) (written : IO.Ref Nat) (record : Json) : IO Unit := do + let line := s!"{marker}{{record.compress}}\n" + let total := (← written.get) + line.utf8ByteSize + if total > probeOutputLimit then + throw <| IO.userError s!"lake env lean exceeded the {{probeOutputLimit}}-byte output limit" + written.set total + out.putStr line + +def kindOf : ConstantInfo → String + | .defnInfo _ => "def" + | .thmInfo _ => "theorem" + | .axiomInfo _ => "axiom" + | .opaqueInfo _ => "opaque" + | .inductInfo _ => "inductive" + | .ctorInfo _ => "constructor" + | .recInfo _ => "recursor" + | .quotInfo _ => "quot" + +/-- The longest proper prefix of the user-facing name that is itself a +constant. A private name's prefixes are tried under its own private prefix +first, so a `where` helper of a private declaration finds that declaration. -/ +def parentOf (env : Environment) (c : Name) : Option Name := Id.run do + let privatePrefix := privatePrefix? c + let mut p := (privateToUserName c).getPrefix + while !p.isAnonymous do + if let some pre := privatePrefix then + if env.contains (pre ++ p) then return some (pre ++ p) + if env.contains p then return some p + p := p.getPrefix + return none + +def nameJson (n : Name) : Json := Json.str (toString n) + +def namesJson (names : Array Name) : Json := + Json.arr ((names.qsort Name.lt).map nameJson) + +/-- What a revision of any project-local constant can reach through `c`: the +local constants its type and its value mention, and its deprecation state. +Theorem values are read too (`allowOpaque`), since proofs break when the +statements they use change; an inductive's constructors stand in for a value. +`internal` is judged on the user-facing name: a private declaration someone +wrote is not an internal detail, while the companions Lean generates for it +(`_proof_1`, `match_1`, `_simp_1`) still are. -/ +def record (env : Environment) (isLocal : Name → Bool) (c : Name) (info : ConstantInfo) + (module : Name) : Json := + let value := info.value? (allowOpaque := true) + let typeConstants := info.type.getUsedConstants + let valueConstants := match info with + | .inductInfo v => v.ctors.toArray + | _ => (value.map (·.getUsedConstants)).getD #[] + let valueMissing := match info with + | .thmInfo _ | .defnInfo _ | .opaqueInfo _ => value.isNone + | _ => false + let usesDeprecated := (typeConstants ++ valueConstants).foldl + (fun acc d => if Linter.isDeprecated env d && !acc.contains d then acc.push d else acc) #[] + Json.mkObj [ + ("name", nameJson c), + ("kind", Json.str (kindOf info)), + ("instance", Json.bool (isInstanceCore env c)), + ("internal", Json.bool (privateToUserName c).isInternalDetail), + ("module", nameJson module), + ("parent", ((parentOf env c).map nameJson).getD Json.null), + ("type_uses", namesJson (typeConstants.filter isLocal)), + ("value_uses", namesJson (valueConstants.filter isLocal)), + ("deprecated", Json.bool (Linter.isDeprecated env c)), + ("replacement", ((Linter.getDeprecatedNewName env c).map nameJson).getD Json.null), + ("uses_deprecated", namesJson usesDeprecated), + ("value_missing", Json.bool valueMissing)] + +end AutoformImpact + +set_option maxHeartbeats 0 in +run_cmd do + let projectRoots : List Name := [{project_roots}] + let env ← getEnv + -- `env.header` is slow to reach from the interpreted probe, so module names + -- are read once. + let moduleNames := env.header.moduleNames + let moduleOf (n : Name) : Option Name := + (env.getModuleIdxFor? n).map fun idx => moduleNames[idx.toNat]! + let isLocalModule (m : Name) : Bool := projectRoots.any (fun projectRoot => projectRoot.isPrefixOf m) + let isLocal (n : Name) : Bool := (moduleOf n).any isLocalModule + -- Records go to the file the CLI names: a command's stdout is buffered into + -- one message, and any other write could split a record. + let some path ← IO.getEnv "{output_env}" | throwError "{output_env} is not set" + let out ← IO.FS.Handle.mk path .write + let written ← IO.mkRef 0 + try + for (c, info) in env.constants do + let some module := moduleOf c | continue + if isLocalModule module then + AutoformImpact.emitRecord out written (AutoformImpact.record env isLocal c info module) + finally + out.flush diff --git a/autoform_cli/skeleton.py b/autoform_cli/skeleton.py index f742796f..9b701448 100644 --- a/autoform_cli/skeleton.py +++ b/autoform_cli/skeleton.py @@ -1186,6 +1186,7 @@ class LeanLibrary: name: str src_dir: Path roots: tuple[str, ...] + globs: tuple[str, ...] = () def lean_libraries(lean_root: str | Path) -> tuple[LeanLibrary, ...]: @@ -1220,7 +1221,14 @@ def lean_libraries(lean_root: str | Path) -> tuple[LeanLibrary, ...]: roots = entry.get("roots") if not isinstance(roots, list) or not all(isinstance(item, str) for item in roots): roots = [name] - libraries.append(LeanLibrary(name=name, src_dir=src_dir.resolve(), roots=tuple(roots))) + # Lake accepts one glob or an array, and translate-config writes them + # only when they differ from the default of one glob per root. + globs = entry.get("globs") + if isinstance(globs, str): + globs = [globs] + if not isinstance(globs, list) or not all(isinstance(item, str) for item in globs): + globs = [] + libraries.append(LeanLibrary(name=name, src_dir=src_dir.resolve(), roots=tuple(roots), globs=tuple(globs))) if not libraries: package = config.get("name") if isinstance(package, str) and package: @@ -1391,11 +1399,13 @@ def run_probe( *, timeout: float = DEFAULT_PROBE_TIMEOUT, freshness_timeout: float = DEFAULT_FRESHNESS_TIMEOUT, + label: str = "skeleton probe", ) -> str: """Run ``probe`` with ``lake env lean`` inside the built project. ``freshness_timeout`` bounds the Lake freshness check that runs first and - ``timeout`` the probe itself; neither spends the other's budget. + ``timeout`` the probe itself; neither spends the other's budget. ``label`` + names the probe in failure messages. """ lake = shutil.which("lake") @@ -1439,12 +1449,12 @@ def run_probe( root = shadowed.group(1).split(".", 1)[0] raise SkeletonError( [ - f"the skeleton probe cannot load toolchain module {shadowed.group(1)}: a dependency " + f"the {label} cannot load toolchain module {shadowed.group(1)}: a dependency " f"library probably provides modules under `{root}`, which hides the toolchain's own `{root}`; " f"rename that library's modules\n{detail}" ] ) - raise SkeletonError([f"the skeleton probe failed; is the project built with `lake build`?\n{detail}"]) + raise SkeletonError([f"the {label} failed; is the project built with `lake build`?\n{detail}"]) return output or result.stdout diff --git a/tests/test_impact.py b/tests/test_impact.py new file mode 100644 index 00000000..1355fc1d --- /dev/null +++ b/tests/test_impact.py @@ -0,0 +1,955 @@ +from __future__ import annotations + +import json +import os +import shutil +import subprocess +from pathlib import Path + +import pytest + +from autoform_cli import __main__ as cli, skeleton +from autoform_cli.impact import ( + IMPACT_MARKER, + IMPACT_SCHEMA, + ConstantRecord, + ImpactArticle, + ImpactError, + compute_impact, + format_impact, + parse_impact_output, + project_modules, + render_impact_probe, + revision_impact, +) +from autoform_cli.skeleton import LeanLibrary, SkeletonError, lean_libraries +from autoform_cli.work import work_context + +_SKELETON_FIXTURE = Path(__file__).resolve().parent / "fixtures" / "skeleton-project" + + +# --------------------------------------------------------------------------- # +# Hand-written records +# --------------------------------------------------------------------------- # + + +def _rec( + name: str, + kind: str = "theorem", + *, + module: str = "Demo", + type_uses: tuple[str, ...] = (), + value_uses: tuple[str, ...] = (), + **fields: object, +) -> ConstantRecord: + return ConstantRecord(name=name, kind=kind, module=module, type_uses=type_uses, value_uses=value_uses, **fields) + + +def _records(*records: ConstantRecord) -> dict[str, ConstantRecord]: + return {record.name: record for record in records} + + +def _article( + node_id: str, *declarations: str, article_id: str | None = None, dependencies: tuple[str, ...] = () +) -> ImpactArticle: + return ImpactArticle(node_id, article_id, declarations, dependencies) + + +def _impact(records, articles, revised: str, declarations=None, **kwargs): + article = next(item for item in articles if item.id == revised) + names = article.declarations if declarations is None else declarations + return compute_impact(records, articles, article, names, source_revision="rev", **kwargs) + + +def _ids(items) -> list[str]: + return [item.id for item in items] + + +def test_type_uses_impact_statements_and_theorem_values_impact_proofs() -> None: + records = _records( + _rec("A.f", "def"), + _rec("A.stated", type_uses=("A.f",)), + _rec("A.proved", value_uses=("A.f",)), + _rec("A.chained", value_uses=("A.stated",)), + _rec("A.further", value_uses=("A.proved",)), + _rec("A.byDef", "def", value_uses=("A.f",)), + _rec("A.byOpaque", "opaque", value_uses=("A.f",)), + _rec("A.overDef", type_uses=("A.byDef",)), + _rec("A.ax", "axiom", type_uses=("A.byOpaque",)), + ) + articles = [ + _article("f", "A.f"), + _article("stated", "A.stated"), + _article("proved", "A.proved"), + _article("chained", "A.chained"), + _article("further", "A.further"), + _article("by-def", "A.byDef"), + _article("by-opaque", "A.byOpaque"), + _article("over-def", "A.overDef"), + _article("ax", "A.ax"), + ] + + report = _impact(records, articles, "f") + + # A definition's or opaque constant's value is part of its meaning, so + # what states anything about them changes too; a theorem's proof is not, + # so proof impact stops one step past the statements it uses. + assert _ids(report.statement_impacted) == ["ax", "by-def", "by-opaque", "over-def", "stated"] + assert _ids(report.proof_impacted) == ["chained", "proved"] + assert report.helpers == () + assert not report.contained + + +def test_constructor_edges_carry_an_inductive_s_meaning() -> None: + records = _records( + _rec("A.Size", "def"), + _rec("A.Shape", "inductive", value_uses=("A.Shape.mk",)), + _rec("A.Shape.mk", "constructor", type_uses=("A.Size", "A.Shape"), parent="A.Shape"), + _rec("A.Shape.rec", "recursor", type_uses=("A.Shape", "A.Shape.mk"), parent="A.Shape"), + _rec("A.area", "def", type_uses=("A.Shape",)), + _rec("A.Other", "inductive", value_uses=("A.Other.mk",)), + _rec("A.Other.mk", "constructor", type_uses=("A.Other",), parent="A.Other"), + ) + articles = [ + _article("size", "A.Size"), + _article("shape", "A.Shape"), + _article("area", "A.area"), + _article("other", "A.Other"), + ] + + report = _impact(records, articles, "size") + + assert _ids(report.statement_impacted) == ["area", "shape"] + assert report.proof_impacted == () + assert [(helper.name, helper.kind, helper.impact, helper.owner) for helper in report.helpers] == [ + ("A.Shape.mk", "constructor", "statement", "shape"), + ("A.Shape.rec", "recursor", "statement", "shape"), + ] + + +def test_simp_companions_and_nested_proofs_carry_proof_impact() -> None: + records = _records( + _rec("A.P", "def"), + _rec("A.P_iff", type_uses=("A.P",)), + _rec("A.P_iff._simp_1", type_uses=("A.P",), value_uses=("A.P_iff",), internal=True, parent="A.P_iff"), + _rec("A.usesSimp", type_uses=("A.P",), value_uses=("A.P_iff._simp_1",)), + _rec("A.g._proof_1", value_uses=("A.P_iff",), internal=True, parent="A.g"), + _rec("A.g", "def", value_uses=("A.g._proof_1",)), + _rec("A.mid", value_uses=("A.P_iff",)), + _rec("A.top", value_uses=("A.mid",)), + ) + articles = [ + _article("p", "A.P"), + _article("iff", "A.P_iff"), + _article("uses-simp", "A.usesSimp"), + _article("g", "A.g"), + _article("top", "A.top"), + ] + + report = _impact(records, articles, "iff") + + # `simp` reaches the lemma only through its `_simp_1` companion, and the + # definition only through the `_proof_1` its nested proof became; a + # theorem someone wrote (A.mid) stops the propagation instead. + assert report.statement_impacted == () + assert _ids(report.proof_impacted) == ["g", "uses-simp"] + assert [(helper.name, helper.impact) for helper in report.helpers] == [("A.mid", "proof")] + + +def test_helpers_report_location_and_the_article_owning_the_nearest_ancestor() -> None: + records = _records( + _rec("A.base", "def"), + _rec("A.base.match_1", "def", type_uses=("A.base",), internal=True, parent="A.base"), + _rec("A.uses", type_uses=("A.base",)), + _rec("A.uses.aux", type_uses=("A.base",), parent="A.uses"), + _rec("A.uses.aux.deep", value_uses=("A.base",), parent="A.uses.aux"), + _rec("A.shared", "def"), + _rec("A.shared.aux", type_uses=("A.base",), parent="A.shared"), + _rec("_private.Demo.Extra.0.A.priv", module="Demo.Extra", value_uses=("A.base",)), + ) + articles = [ + _article("base", "A.base"), + _article("uses", "A.uses"), + _article("shared-b", "A.shared"), + _article("shared-a", "A.shared"), + ] + located: list[str] = [] + + def locate(record: ConstantRecord) -> tuple[str | None, int | None]: + located.append(record.name) + return ("Demo.lean", len(located)) if record.module == "Demo" else (None, None) + + report = _impact(records, articles, "base", locate=locate) + + assert [helper.as_dict() for helper in report.helpers] == [ + { + "name": "A.shared.aux", + "kind": "theorem", + "impact": "statement", + "module": "Demo", + "path": "Demo.lean", + "line": 1, + "owner": "shared-a", + }, + { + "name": "A.uses.aux", + "kind": "theorem", + "impact": "statement", + "module": "Demo", + "path": "Demo.lean", + "line": 2, + "owner": "uses", + }, + { + "name": "A.uses.aux.deep", + "kind": "theorem", + "impact": "proof", + "module": "Demo", + "path": "Demo.lean", + "line": 3, + "owner": "uses", + }, + { + "name": "_private.Demo.Extra.0.A.priv", + "kind": "theorem", + "impact": "proof", + "module": "Demo.Extra", + "path": None, + "line": None, + "owner": None, + }, + ] + assert located == [helper.name for helper in report.helpers] + assert _ids(report.statement_impacted) == ["uses"] + + +def test_revised_names_resolve_by_component_and_are_never_their_own_helpers() -> None: + records = _records(_rec("A.leaf", "def"), _rec("A.user", type_uses=("A.leaf",))) + articles = [_article("owner", "A.user"), _article("empty")] + + report = _impact(records, articles, "empty", ["A.«leaf»", "A.leaf"]) + + assert report.declarations == ("A.«leaf»",) + assert _ids(report.statement_impacted) == ["owner"] + assert report.helpers == () + + +def test_revising_names_that_are_not_project_local_is_refused() -> None: + records = _records(_rec("A.leaf", "def")) + articles = [_article("leaf", "A.leaf")] + + with pytest.raises(ImpactError, match=r"^not a project-local constant: Nat\.add, A\.missing$"): + _impact(records, articles, "leaf", ["Nat.add", "A.leaf", "A.missing", "Nat.add"]) + with pytest.raises(ImpactError, match="^leaf: nothing to revise; pass --declaration NAME$"): + _impact(records, articles, "leaf", []) + + +def test_undeclared_dependencies_and_claim_targets() -> None: + records = _records( + _rec("A.base", "def"), + _rec("A.direct", type_uses=("A.base",)), + _rec("A.transitive", type_uses=("A.base",)), + _rec("A.loose", value_uses=("A.base",)), + _rec("A.detached", type_uses=("A.base",)), + ) + articles = [ + _article("chapter/base", "A.base", article_id="af_base"), + _article("chapter/direct", "A.direct", article_id="af_direct", dependencies=("chapter/base",)), + _article("chapter/transitive", "A.transitive", dependencies=("chapter/direct",)), + _article("chapter/loose", "A.loose", article_id="af_loose", dependencies=("chapter/middle",)), + _article("chapter/middle", dependencies=("chapter/loose",)), + _article("chapter/detached", "A.detached", article_id="af_detached"), + ] + + report = _impact(records, articles, "chapter/base") + + assert _ids(report.statement_impacted) == ["chapter/detached", "chapter/direct", "chapter/transitive"] + assert _ids(report.proof_impacted) == ["chapter/loose"] + assert report.undeclared_dependencies == ("chapter/detached", "chapter/loose") + assert report.claim_targets == ("af_base", "af_detached", "af_direct", "af_loose", "chapter/transitive") + + +def test_deprecated_constants_report_users_through_internal_details() -> None: + records = _records( + _rec("A.new"), + _rec("A.old", deprecated=True, replacement="A.new"), + _rec("A.user", value_uses=("A.old",), uses_deprecated=("A.old",)), + _rec( + "A.wrapped._proof_1", + value_uses=("A.old",), + uses_deprecated=("A.old",), + internal=True, + parent="A.wrapped", + ), + _rec("A.wrapped", "def", value_uses=("A.wrapped._proof_1",)), + _rec("A.older", deprecated=True), + _rec("A.stray._proof_1", value_uses=("A.older",), uses_deprecated=("A.older",), internal=True), + _rec("A.unused", deprecated=True, replacement="A.new"), + ) + articles = [_article("new", "A.new")] + + report = _impact(records, articles, "new") + + # An internal user stands for its own users, unless nothing uses it: a + # constant something still mentions is never reported as safe to delete. + assert [item.as_dict() for item in report.deprecated] == [ + {"name": "A.old", "replacement": "A.new", "users": ["A.user", "A.wrapped"]}, + {"name": "A.older", "replacement": None, "users": ["A.stray._proof_1"]}, + {"name": "A.unused", "replacement": "A.new", "users": []}, + ] + assert report.deprecated_unused == ("A.unused",) + assert report.contained + + +def test_a_contained_revision_says_it_can_be_revised_in_place() -> None: + records = _records(_rec("A.leaf", "def"), _rec("A.other")) + articles = [_article("chapter/leaf", "A.leaf", article_id="af_leaf"), _article("chapter/other", "A.other")] + + report = _impact(records, articles, "chapter/leaf") + + assert report.contained + assert report.as_dict()["contained"] is True + assert report.claim_targets == ("af_leaf",) + assert format_impact(report) == [ + "Revising A.leaf of chapter/leaf [af_leaf]", + "Graph source revision: rev", + "Contained: no other article or helper uses A.leaf, so it can be revised in place.", + "Claim targets: af_leaf", + ] + + +# --------------------------------------------------------------------------- # +# Probe output +# --------------------------------------------------------------------------- # + + +def _payload(name: str, **fields: object) -> dict[str, object]: + record: dict[str, object] = { + "name": name, + "kind": "theorem", + "instance": False, + "internal": False, + "module": "Demo", + "parent": None, + "type_uses": [], + "value_uses": [], + "deprecated": False, + "replacement": None, + "uses_deprecated": [], + "value_missing": False, + } + record.update(fields) + return record + + +def _line(payload: dict[str, object]) -> str: + return IMPACT_MARKER + json.dumps(payload) + + +def test_probe_output_is_read_from_marker_lines() -> None: + text = "\n".join( + [ + "warning: unrelated Lean output", + _line(_payload("A.b", type_uses=["A.a"], parent="A")), + _line(_payload("A.a", kind="def", instance=True)), + "", + ] + ) + + records = parse_impact_output(text) + + assert records == { + "A.a": ConstantRecord(name="A.a", kind="def", module="Demo", instance=True), + "A.b": ConstantRecord(name="A.b", kind="theorem", module="Demo", type_uses=("A.a",), parent="A"), + } + + +@pytest.mark.parametrize( + ("lines", "message"), + [ + ([IMPACT_MARKER + "{"], "emitted invalid JSON"), + ([_line({**_payload("A.a"), "extra": 1})], "record with unexpected fields"), + ([_line(_payload("A.a", type_uses="A.b"))], "malformed 'type_uses' field"), + ([_line(_payload("A.a", value_uses=[1]))], "malformed 'value_uses' field"), + ([_line(_payload("A.a", kind="lemma"))], "malformed record for 'A.a'"), + ([_line(_payload(""))], "malformed record for ''"), + ([_line(_payload("A.a")), _line(_payload("A.a"))], "emitted A.a twice"), + (["no records here"], "found no project-local constants"), + ([_line(_payload("A.a", value_missing=True))], "could not read the value of A.a"), + ([_line(_payload("A.a", value_uses=["A.gone"]))], r"A\.a using unknown constants: \['A\.gone'\]"), + ], +) +def test_malformed_probe_output_fails_closed(lines: list[str], message: str) -> None: + with pytest.raises(SkeletonError, match=message): + parse_impact_output("\n".join(lines)) + + +def test_rendered_probe_imports_modules_and_names_local_prefixes() -> None: + source = render_impact_probe(imports=["Demo.B", "Demo", "Demo.B"], project_roots=["Demo.B", "Demo"]) + + assert source.startswith("import Demo\nimport Demo.B\n-- Autoform impact probe.") + assert ( + 'let projectRoots : List Name := [Name.str (Name.anonymous) "Demo", ' + 'Name.str (Name.str (Name.anonymous) "Demo") "B"]' + ) in source + assert f'"{skeleton.PROBE_OUTPUT_ENV}"' in source + assert IMPACT_MARKER in source + with pytest.raises(SkeletonError, match="no imports"): + render_impact_probe(imports=[], project_roots=["Demo"]) + + +def test_probe_failures_name_the_impact_probe(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + (tmp_path / "lake-manifest.json").write_text("{}\n", encoding="utf-8") + monkeypatch.setattr("autoform_cli.skeleton.shutil.which", lambda executable: "/bin/lake") + monkeypatch.setattr("autoform_cli.skeleton._check_artifacts_fresh", lambda *args, **kwargs: None) + monkeypatch.setattr( + "autoform_cli.skeleton._run_bounded_command", + lambda command, **kwargs: subprocess.CompletedProcess(command, 1, stdout="", stderr="unknown module"), + ) + probe = render_impact_probe(imports=["Demo"], project_roots=["Demo"]) + + with pytest.raises(SkeletonError) as impact: + skeleton.run_probe(probe, tmp_path, label="impact probe") + with pytest.raises(SkeletonError) as default: + skeleton.run_probe(probe, tmp_path) + + assert impact.value.issues == ("the impact probe failed; is the project built with `lake build`?\nunknown module",) + assert default.value.issues == ( + "the skeleton probe failed; is the project built with `lake build`?\nunknown module", + ) + + +# --------------------------------------------------------------------------- # +# Project modules +# --------------------------------------------------------------------------- # + + +def _sources(root: Path, *modules: str) -> Path: + for module in modules: + path = root.joinpath(*module.split(".")).with_suffix(".lean") + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text("", encoding="utf-8") + return root + + +def _library(src: Path, *globs: str, roots: tuple[str, ...] = ("Demo",)) -> LeanLibrary: + return LeanLibrary(name="Demo", src_dir=src, roots=roots, globs=globs) + + +def test_project_modules_follow_lake_globs(tmp_path: Path) -> None: + src = _sources( + tmp_path, + "Demo", + "Demo.Sub", + "Demo.Sub.A", + "Demo.Sub.B.C", + "Demo.Extra", + "Demo.Extra.X", + "Demo.Extra.Y.Z", + "Other.Lone", + ) + (tmp_path / "Demo" / "Sub" / "notes.md").write_text("", encoding="utf-8") + + # `M` is one module, `M.*` the module and its submodules, `M.+` only its + # submodules; every glob base marks local constants, like a root. + assert project_modules([_library(src, "Demo", "Demo.Sub.*", "Demo.Extra.+")]) == ( + ("Demo", "Demo.Extra.X", "Demo.Extra.Y.Z", "Demo.Sub", "Demo.Sub.A", "Demo.Sub.B.C"), + ("Demo", "Demo.Extra", "Demo.Sub"), + ) + # Without globs, Lake builds the roots. + assert project_modules([_library(src, roots=("Demo", "Other.Lone"))]) == ( + ("Demo", "Other.Lone"), + ("Demo", "Other.Lone"), + ) + + +@pytest.mark.parametrize( + ("glob", "message"), + [ + ("Demo.**", r"cannot read module glob 'Demo\.\*\*'"), + ("Demo/Odd", "cannot read module glob 'Demo/Odd'"), + ("Demo.Missing", "module Demo.Missing has no source file"), + ("Demo.Missing.*", "module Demo.Missing has no source file"), + ("Demo.Missing.+", r"glob Demo\.Missing\.\+ names no source directory Demo/Missing"), + ("Demo.Odd.+", r"cannot import .*bad-name\.lean"), + ("Demo.Empty.+", "the Lake configuration selects no modules"), + ], +) +def test_project_modules_fail_closed(tmp_path: Path, glob: str, message: str) -> None: + src = _sources(tmp_path, "Demo", "Demo.Odd.Fine") + (tmp_path / "Demo" / "Odd" / "bad-name.lean").write_text("", encoding="utf-8") + (tmp_path / "Demo" / "Empty").mkdir() + + with pytest.raises(SkeletonError, match=message): + project_modules([_library(src, glob)]) + + +def test_lean_libraries_read_one_glob_or_an_array(tmp_path: Path) -> None: + (tmp_path / "lakefile.toml").write_text( + 'name = "Demo"\n\n' + '[[lean_lib]]\nname = "One"\nglobs = "One.+"\n\n' + '[[lean_lib]]\nname = "Many"\nglobs = ["Many", "Many.Sub.*"]\n\n' + '[[lean_lib]]\nname = "Plain"\n\n' + '[[lean_lib]]\nname = "Odd"\nglobs = [1]\n', + encoding="utf-8", + ) + + assert {library.name: (library.roots, library.globs) for library in lean_libraries(tmp_path)} == { + "One": (("One",), ("One.+",)), + "Many": (("Many",), ("Many", "Many.Sub.*")), + "Plain": (("Plain",), ()), + "Odd": (("Odd",), ()), + } + + +# --------------------------------------------------------------------------- # +# The command, with a stubbed probe +# --------------------------------------------------------------------------- # + +_BASE_ID = "af_000000000000000000000001" +_USES_ID = "af_000000000000000000000002" + + +def _write_article( + project: Path, name: str, *, metadata: list[str], depends: tuple[str, ...] = (), title: str | None = None +) -> None: + path = project / "blueprint/roadmap/chapter" / name + path.parent.mkdir(parents=True, exist_ok=True) + text = ["---", *metadata, "---", "", f"# {title or path.stem.title()}", "", "A precise statement."] + if depends: + text.extend(["", "## Depends on", "", *(f"- [dependency]({target})" for target in depends)]) + path.write_text("\n".join(text) + "\n", encoding="utf-8") + + +def _blueprint_project(tmp_path: Path, prefix: str) -> Path: + """A roadmap whose articles name ``{prefix}.base``, ``.uses``, ``.loose`` and nothing.""" + + project = tmp_path / "project" + _write_article(project, "README.md", title="Chapter", metadata=["article_id: af_0000000000000000000000c0"]) + _write_article( + project, + "base.md", + metadata=[f"article_id: {_BASE_ID}", "declaration: definition", "statement: formalized", f"lean: {prefix}.base"], + ) + _write_article( + project, + "uses.md", + metadata=[f"article_id: {_USES_ID}", "declaration: theorem", "statement: formalized", f"lean: {prefix}.uses"], + depends=("base.md",), + ) + _write_article( + project, + "loose.md", + metadata=["declaration: theorem", "statement: formalized", f"lean: {prefix}.loose"], + ) + _write_article(project, "empty.md", metadata=["declaration: theorem"]) + return project + + +def _stub_lean_root(tmp_path: Path) -> Path: + root = tmp_path / "lean" + root.mkdir() + (root / "lakefile.toml").write_text('name = "Demo"\n\n[[lean_lib]]\nname = "Demo"\n', encoding="utf-8") + (root / "Demo.lean").write_text( + "namespace Demo\n\ndef base (n : Nat) : Nat := n\n\ntheorem base_eq (n : Nat) : base n = n := rfl\n\nend Demo\n", + encoding="utf-8", + ) + return root + + +_STUB_RECORDS = [ + _payload("Demo.base", kind="def"), + _payload("Demo.uses", type_uses=["Demo.base"]), + _payload("Demo.base_eq", type_uses=["Demo.base"]), + _payload("Demo.loose", value_uses=["Demo.base", "Demo.old"], uses_deprecated=["Demo.old"]), + _payload("Demo.old", deprecated=True, replacement="Demo.base_eq"), + _payload("Demo.gone", deprecated=True), +] + + +def _stub_probe(monkeypatch: pytest.MonkeyPatch, records=_STUB_RECORDS, *, error: SkeletonError | None = None): + calls: list[dict[str, object]] = [] + + def run_probe(probe: str, lean_root: Path, **kwargs: object) -> str: + calls.append({"probe": probe, "lean_root": lean_root, **kwargs}) + if error is not None: + raise error + return "\n".join(["lake env lean noise", *(_line(record) for record in records)]) + "\n" + + monkeypatch.setattr("autoform_cli.skeleton.run_probe", run_probe) + return calls + + +def test_cli_writes_the_impact_report_as_canonical_json(tmp_path: Path, monkeypatch, capsys) -> None: + project = _blueprint_project(tmp_path, "Demo") + lean_root = _stub_lean_root(tmp_path) + calls = _stub_probe(monkeypatch) + source_revision, _ = work_context(project, "chapter/base") + + code = cli.main( + ["work", "impact", _BASE_ID, str(project), "--lean-root", str(lean_root), "--json", "--timeout", "30"] + ) + + output = capsys.readouterr() + assert code == 0, output.err + assert output.err == "" + report = json.loads(output.out) + assert output.out == json.dumps(report, sort_keys=True, separators=(",", ":")) + "\n" + assert report == { + "schema": IMPACT_SCHEMA, + "source_revision": source_revision, + "article": {"id": "chapter/base", "article_id": _BASE_ID, "claim_target": _BASE_ID}, + "declarations": ["Demo.base"], + "contained": False, + "statement_impacted": [ + {"id": "chapter/uses", "article_id": _USES_ID, "claim_target": _USES_ID, "declarations": ["Demo.uses"]} + ], + "proof_impacted": [ + {"id": "chapter/loose", "article_id": None, "claim_target": "chapter/loose", "declarations": ["Demo.loose"]} + ], + "helpers": [ + { + "name": "Demo.base_eq", + "kind": "theorem", + "impact": "statement", + "module": "Demo", + "path": "Demo.lean", + "line": 5, + "owner": None, + } + ], + "undeclared_dependencies": ["chapter/loose"], + "deprecated": [ + {"name": "Demo.gone", "replacement": None, "users": []}, + {"name": "Demo.old", "replacement": "Demo.base_eq", "users": ["Demo.loose"]}, + ], + "deprecated_unused": ["Demo.gone"], + "claim_targets": [_BASE_ID, _USES_ID, "chapter/loose"], + } + (call,) = calls + assert call["label"] == "impact probe" + assert call["timeout"] == 30 + assert call["lean_root"] == lean_root.resolve() + assert str(call["probe"]).startswith("import Demo\n-- Autoform impact probe.") + + +def test_cli_text_report_lists_each_section(tmp_path: Path, monkeypatch, capsys) -> None: + project = _blueprint_project(tmp_path, "Demo") + lean_root = _stub_lean_root(tmp_path) + calls = _stub_probe(monkeypatch) + source_revision, _ = work_context(project, "chapter/base") + + code = cli.main(["work", "impact", "chapter/base", str(project), "--lean-root", str(lean_root)]) + + output = capsys.readouterr() + assert code == 0, output.err + assert output.out.splitlines() == [ + f"Revising Demo.base of chapter/base [{_BASE_ID}]", + f"Graph source revision: {source_revision}", + "Statement impacted:", + f" chapter/uses [{_USES_ID}]: Demo.uses", + "Proof impacted:", + " chapter/loose: Demo.loose", + "Helpers no article names:", + " Demo.base_eq (theorem, statement) Demo.lean:5; no owner", + "Impacted without a Markdown dependency path to the revised article: chapter/loose", + "Deprecated:", + " Demo.gone: no users, safe to delete", + " Demo.old -> Demo.base_eq: used by Demo.loose", + f"Claim targets: {_BASE_ID}, {_USES_ID}, chapter/loose", + ] + assert calls[0]["timeout"] == skeleton.DEFAULT_PROBE_TIMEOUT + + +def test_cli_declaration_flag_replaces_the_article_s_names(tmp_path: Path, monkeypatch, capsys) -> None: + project = _blueprint_project(tmp_path, "Demo") + lean_root = _stub_lean_root(tmp_path) + _stub_probe(monkeypatch) + + code = cli.main( + [ + "work", + "impact", + "chapter/empty", + str(project), + "--lean-root", + str(lean_root), + "--declaration", + "Demo.gone", + "--declaration", + "Demo.gone", + "--json", + ] + ) + + output = capsys.readouterr() + assert code == 0, output.err + report = json.loads(output.out) + assert report["article"] == {"id": "chapter/empty", "article_id": None, "claim_target": "chapter/empty"} + assert report["declarations"] == ["Demo.gone"] + assert report["contained"] is True + assert report["claim_targets"] == ["chapter/empty"] + + +def test_cli_text_escapes_terminal_control_characters(tmp_path: Path, monkeypatch, capsys) -> None: + project = _blueprint_project(tmp_path, "Demo") + lean_root = _stub_lean_root(tmp_path) + _stub_probe(monkeypatch, [*_STUB_RECORDS, _payload("Demo.bad\x1b[2Jname", type_uses=["Demo.base"])]) + + code = cli.main(["work", "impact", "chapter/base", str(project), "--lean-root", str(lean_root)]) + + output = capsys.readouterr() + assert code == 0, output.err + assert "\x1b" not in output.out + assert " Demo.bad\\x1b[2Jname (theorem, statement) Demo.lean; no owner" in output.out.splitlines() + + +@pytest.mark.parametrize( + ("arguments", "message", "probed"), + [ + (["chapter/nope"], "error: no article matches 'chapter/nope'", False), + (["chapter/empty"], "error: chapter/empty names no lean: declaration; pass --declaration NAME", False), + ( + ["chapter/base", "--declaration", "Nat.add", "--declaration", "Demo.\x07bell"], + "error: not a project-local constant: Nat.add, Demo.\\x07bell", + True, + ), + ], +) +def test_cli_refuses_questions_it_cannot_answer( + tmp_path: Path, monkeypatch, capsys, arguments: list[str], message: str, probed: bool +) -> None: + project = _blueprint_project(tmp_path, "Demo") + lean_root = _stub_lean_root(tmp_path) + calls = _stub_probe(monkeypatch) + selector, *flags = arguments + + code = cli.main(["work", "impact", selector, str(project), "--lean-root", str(lean_root), *flags]) + + output = capsys.readouterr() + assert code == 2 + assert output.out == "" + assert output.err.splitlines()[0].startswith(message) + assert bool(calls) is probed + + +def test_cli_reports_probe_failures_line_by_line(tmp_path: Path, monkeypatch, capsys) -> None: + project = _blueprint_project(tmp_path, "Demo") + lean_root = _stub_lean_root(tmp_path) + failure = SkeletonError(["the impact probe failed; is the project built with `lake build`?\nDemo.lean:1:0: \x1b[31m"]) + _stub_probe(monkeypatch, error=failure) + + code = cli.main(["work", "impact", "chapter/base", str(project), "--lean-root", str(lean_root)]) + + output = capsys.readouterr() + assert code == 2 + assert output.out == "" + assert output.err.splitlines() == [ + "error: the impact probe failed; is the project built with `lake build`?", + "Demo.lean:1:0: \\x1b[31m", + ] + + +def test_cli_requires_a_lake_project_and_a_lean_root(tmp_path: Path, monkeypatch, capsys) -> None: + project = _blueprint_project(tmp_path, "Demo") + empty_root = tmp_path / "not-lean" + empty_root.mkdir() + calls = _stub_probe(monkeypatch) + + code = cli.main(["work", "impact", "chapter/base", str(project), "--lean-root", str(empty_root)]) + + output = capsys.readouterr() + assert code == 2 + assert output.err.splitlines() == [f"error: no lakefile.toml or lakefile.lean in {empty_root.resolve()}"] + assert calls == [] + with pytest.raises(SystemExit) as missing: + cli.main(["work", "impact", "chapter/base", str(project)]) + assert missing.value.code == 2 + + +def test_revision_impact_reports_a_roadmap_without_the_selected_article(tmp_path: Path, monkeypatch) -> None: + project = _blueprint_project(tmp_path, "Demo") + lean_root = _stub_lean_root(tmp_path) + _stub_probe(monkeypatch) + + report = revision_impact(project, "chapter/uses", lean_root=lean_root) + + assert report.declarations == ("Demo.uses",) + assert report.contained + assert report.claim_targets == (_USES_ID,) + + +# --------------------------------------------------------------------------- # +# The real probe +# --------------------------------------------------------------------------- # + + +def _lean_toolchain_available() -> bool: + """Whether a project pinned like the skeleton fixture builds here without a download. + + This repeats tests/test_skeleton.py's check, including its + AUTOFORM_REQUIRE_REAL_LEAN_TESTS switch, rather than importing that module. + """ + + def unavailable(reason: str) -> bool: + if os.environ.get("AUTOFORM_REQUIRE_REAL_LEAN_TESTS") == "1": + raise RuntimeError(f"real Lean tests are required but unavailable: {reason}") + return False + + if shutil.which("lake") is None: + return unavailable("lake is not on PATH") + elan = shutil.which("elan") + if elan is None: + return True + pinned = (_SKELETON_FIXTURE / "lean-toolchain").read_text(encoding="utf-8").strip() + try: + listed = subprocess.run([elan, "toolchain", "list"], capture_output=True, text=True, timeout=30, check=False) + except (OSError, subprocess.SubprocessError): + return unavailable("elan toolchain discovery failed") + available = any(line.split()[:1] == [pinned] for line in listed.stdout.splitlines()) + return available or unavailable(f"{pinned} is not installed") + + +_IMP_BASIC = """\ +namespace Imp + +def base (n : Nat) : Nat := n + +theorem base_eq (n : Nat) : base n = n := rfl + +theorem uses (n : Nat) : base n = n := base_eq n + +def P (n : Nat) : Prop := base n = n + +@[simp] theorem P_iff (n : Nat) : P n ↔ True := ⟨fun _ => trivial, fun _ => base_eq n⟩ + +@[deprecated base_eq (since := "2026-10-05")] +theorem oldEq (n : Nat) : base n = n := base_eq n + +@[deprecated base_eq (since := "2026-10-05")] +theorem oldUnused (n : Nat) : base n = n := base_eq n + +end Imp +""" + +_IMP_EXTRA = """\ +import Imp.Basic + +namespace Imp + +theorem usesSimp (n : Nat) : P n := by simp + +theorem proved (n : Nat) : n + 0 = n := base_eq n + +set_option linter.deprecated false in +theorem usesOld (n : Nat) : base n = n := oldEq n + +theorem apart (n : Nat) : n = n := rfl + +end Imp +""" + + +def _imp_project(root: Path) -> Path: + """A built Lean project whose glob selects a module its root never imports.""" + + lean_root = root / "lean" + (lean_root / "Imp").mkdir(parents=True) + shutil.copy(_SKELETON_FIXTURE / "lean-toolchain", lean_root / "lean-toolchain") + (lean_root / "lakefile.toml").write_text( + 'name = "Imp"\ndefaultTargets = ["Imp"]\n\n[[lean_lib]]\nname = "Imp"\nglobs = ["Imp.*"]\n', + encoding="utf-8", + ) + (lean_root / "Imp.lean").write_text("import Imp.Basic\n", encoding="utf-8") + (lean_root / "Imp" / "Basic.lean").write_text(_IMP_BASIC, encoding="utf-8") + (lean_root / "Imp" / "Extra.lean").write_text(_IMP_EXTRA, encoding="utf-8") + build = subprocess.run(["lake", "build"], cwd=lean_root, capture_output=True, text=True, timeout=600, check=False) + assert build.returncode == 0, build.stdout + build.stderr + return lean_root + + +def _imp_roadmap(root: Path) -> Path: + project = root / "project" + _write_article(project, "README.md", title="Chapter", metadata=["article_id: af_0000000000000000000000c0"]) + for name, declaration, lean, depends in ( + ("base.md", "definition", "Imp.base", ()), + ("uses.md", "theorem", "Imp.uses", ("base.md",)), + ("simp.md", "theorem", "Imp.usesSimp", ()), + ("proved.md", "theorem", "Imp.proved", ("uses.md",)), + ("apart.md", "theorem", "Imp.apart", ()), + ): + metadata = [f"declaration: {declaration}", "statement: formalized", f"lean: {lean}"] + _write_article(project, name, metadata=metadata, depends=depends) + return project + + +@pytest.mark.skipif(not _lean_toolchain_available(), reason="needs lake and the fixture's Lean toolchain") +def test_the_probe_reads_a_built_project(tmp_path: Path, monkeypatch, capsys) -> None: + lean_root = _imp_project(tmp_path) + project = _imp_roadmap(tmp_path) + source_revision, _ = work_context(project, "chapter/base") + outputs: list[str] = [] + run_probe = skeleton.run_probe + + def recorded(*args: object, **kwargs: object) -> str: + outputs.append(run_probe(*args, **kwargs)) + return outputs[-1] + + monkeypatch.setattr("autoform_cli.skeleton.run_probe", recorded) + + code = cli.main(["work", "impact", "chapter/base", str(project), "--lean-root", str(lean_root), "--json"]) + + output = capsys.readouterr() + assert code == 0, output.err + report = json.loads(output.out) + assert report["declarations"] == ["Imp.base"] + assert report["contained"] is False + # Imp.Extra is reached only through the glob. `simp` closes usesSimp, but + # its statement mentions the definition P, whose value mentions base. + assert [(item["id"], item["declarations"]) for item in report["statement_impacted"]] == [ + ("chapter/simp", ["Imp.usesSimp"]), + ("chapter/uses", ["Imp.uses"]), + ] + assert [(item["id"], item["declarations"]) for item in report["proof_impacted"]] == [ + ("chapter/proved", ["Imp.proved"]) + ] + assert [(h["name"], h["kind"], h["impact"], h["path"], h["line"], h["owner"]) for h in report["helpers"]] == [ + ("Imp.P", "def", "statement", "Imp/Basic.lean", 9, None), + ("Imp.P_iff", "theorem", "statement", "Imp/Basic.lean", 11, None), + ("Imp.base_eq", "theorem", "statement", "Imp/Basic.lean", 5, None), + ("Imp.oldEq", "theorem", "statement", "Imp/Basic.lean", 14, None), + ("Imp.oldUnused", "theorem", "statement", "Imp/Basic.lean", 17, None), + ("Imp.usesOld", "theorem", "statement", "Imp/Extra.lean", 10, None), + ] + assert report["undeclared_dependencies"] == ["chapter/simp"] + assert report["deprecated"] == [ + {"name": "Imp.oldEq", "replacement": "Imp.base_eq", "users": ["Imp.usesOld"]}, + {"name": "Imp.oldUnused", "replacement": "Imp.base_eq", "users": []}, + ] + assert report["deprecated_unused"] == ["Imp.oldUnused"] + assert report["claim_targets"] == ["chapter/base", "chapter/proved", "chapter/simp", "chapter/uses"] + assert report["source_revision"] == source_revision + + (probed,) = outputs + records = parse_impact_output(probed) + assert {record.module for record in records.values()} == {"Imp.Basic", "Imp.Extra"} + simp_lemma = next(record for record in records.values() if record.parent == "Imp.P_iff") + assert simp_lemma.internal and simp_lemma.type_uses == ("Imp.P",) + assert records["Imp.usesOld"].uses_deprecated == ("Imp.oldEq",) + + # The same records answer the other questions without another probe run. + articles = [ + ImpactArticle("chapter/base", None, ("Imp.base",)), + ImpactArticle("chapter/uses", None, ("Imp.uses",), ("chapter/base",)), + ImpactArticle("chapter/simp", None, ("Imp.usesSimp",)), + ImpactArticle("chapter/proved", None, ("Imp.proved",), ("chapter/uses",)), + ImpactArticle("chapter/apart", None, ("Imp.apart",)), + ] + simp_only = compute_impact(records, articles, articles[0], ["Imp.P_iff"], source_revision="rev") + assert simp_only.statement_impacted == () + assert _ids(simp_only.proof_impacted) == ["chapter/simp"] + assert simp_only.helpers == () + assert not simp_only.contained + apart = compute_impact(records, articles, articles[4], ["Imp.apart"], source_revision="rev") + assert apart.contained + assert apart.claim_targets == ("chapter/apart",) From 18c31a3d02f3f70351399f0d367d1c23a52106f9 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 06:27:14 -0400 Subject: [PATCH 08/38] Test impact edge cases and deprecated attributes of earlier declarations Locate a private helper by its source name, and fall back to the module's file without a line when the index finds the name in another file. Refuse a roadmap that changes between work_context and the runtime graph, both when an article is rewritten and when the selected article vanishes. Check that the shadowed-Std probe failure names the impact probe too, and rename the contained-revision test after what it checks. Keep an earlier declaration's @[deprecated], on its own line or before an alias, from marking the theorem that follows it. --- tests/test_audit.py | 10 +++++ tests/test_impact.py | 93 ++++++++++++++++++++++++++++++++++++++++---- 2 files changed, 96 insertions(+), 7 deletions(-) diff --git a/tests/test_audit.py b/tests/test_audit.py index 197f12bd..7d576fd2 100644 --- a/tests/test_audit.py +++ b/tests/test_audit.py @@ -263,6 +263,8 @@ def test_audit_reports_lean_targets_declared_deprecated(tmp_path: Path) -> None: "multi-line.md": "Project.multiLine", "fresh.md": "Project.fresh", "decoy.md": "Project.decoy", + "after-previous.md": "Project.afterPrevious", + "after-alias.md": "Project.afterAlias", } for relative, name in targets.items(): _article(blueprint, relative, declaration="theorem", statement="formalized", lean=name) @@ -289,6 +291,14 @@ def test_audit_reports_lean_targets_declared_deprecated(tmp_path: Path) -> None: "theorem Project.decoy : True := by", ' have : "@[deprecated]" = "@[deprecated]" := rfl', " trivial", + "", + # An earlier declaration's attributes stay with it, indexed or not. + "@[deprecated Project.fresh] theorem Project.previous : True := trivial", + "theorem Project.afterPrevious : True := trivial", + "", + "@[deprecated Project.fresh]", + "alias Project.oldAlias := Project.fresh", + "theorem Project.afterAlias : True := trivial", ] (lean_root / "Old.lean").write_text("\n".join(source) + "\n", encoding="utf-8") diff --git a/tests/test_impact.py b/tests/test_impact.py index 1355fc1d..bcd7b3d6 100644 --- a/tests/test_impact.py +++ b/tests/test_impact.py @@ -22,6 +22,7 @@ render_impact_probe, revision_impact, ) +from autoform_cli.runtime import load_runtime_graph from autoform_cli.skeleton import LeanLibrary, SkeletonError, lean_libraries from autoform_cli.work import work_context @@ -398,13 +399,29 @@ def test_rendered_probe_imports_modules_and_names_local_prefixes() -> None: render_impact_probe(imports=[], project_roots=["Demo"]) -def test_probe_failures_name_the_impact_probe(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: +_SHADOWED_STD = "object file '/deps/Std/Data.olean' of module Std.Data does not exist" + + +@pytest.mark.parametrize( + ("stderr", "message"), + [ + ("unknown module", "the {label} failed; is the project built with `lake build`?\nunknown module"), + ( + _SHADOWED_STD, + "the {label} cannot load toolchain module Std.Data: a dependency library probably provides modules " + "under `Std`, which hides the toolchain's own `Std`; rename that library's modules\n" + _SHADOWED_STD, + ), + ], +) +def test_probe_failures_name_the_impact_probe( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, stderr: str, message: str +) -> None: (tmp_path / "lake-manifest.json").write_text("{}\n", encoding="utf-8") monkeypatch.setattr("autoform_cli.skeleton.shutil.which", lambda executable: "/bin/lake") monkeypatch.setattr("autoform_cli.skeleton._check_artifacts_fresh", lambda *args, **kwargs: None) monkeypatch.setattr( "autoform_cli.skeleton._run_bounded_command", - lambda command, **kwargs: subprocess.CompletedProcess(command, 1, stdout="", stderr="unknown module"), + lambda command, **kwargs: subprocess.CompletedProcess(command, 1, stdout="", stderr=stderr), ) probe = render_impact_probe(imports=["Demo"], project_roots=["Demo"]) @@ -413,10 +430,8 @@ def test_probe_failures_name_the_impact_probe(tmp_path: Path, monkeypatch: pytes with pytest.raises(SkeletonError) as default: skeleton.run_probe(probe, tmp_path) - assert impact.value.issues == ("the impact probe failed; is the project built with `lake build`?\nunknown module",) - assert default.value.issues == ( - "the skeleton probe failed; is the project built with `lake build`?\nunknown module", - ) + assert impact.value.issues == (message.format(label="impact probe"),) + assert default.value.issues == (message.format(label="skeleton probe"),) # --------------------------------------------------------------------------- # @@ -767,7 +782,7 @@ def test_cli_requires_a_lake_project_and_a_lean_root(tmp_path: Path, monkeypatch assert missing.value.code == 2 -def test_revision_impact_reports_a_roadmap_without_the_selected_article(tmp_path: Path, monkeypatch) -> None: +def test_revision_impact_reports_an_article_nothing_uses_as_contained(tmp_path: Path, monkeypatch) -> None: project = _blueprint_project(tmp_path, "Demo") lean_root = _stub_lean_root(tmp_path) _stub_probe(monkeypatch) @@ -779,6 +794,70 @@ def test_revision_impact_reports_a_roadmap_without_the_selected_article(tmp_path assert report.claim_targets == (_USES_ID,) +def test_helpers_are_located_by_source_name_in_their_module_s_file(tmp_path: Path, monkeypatch) -> None: + project = _blueprint_project(tmp_path, "Demo") + lean_root = _stub_lean_root(tmp_path) + (lean_root / "Demo").mkdir() + (lean_root / "Demo" / "Extra.lean").write_text( + "namespace Demo\n\nprivate theorem secret (n : Nat) : base n = n := rfl\n\nend Demo\n", encoding="utf-8" + ) + # A file outside the library that reuses names the library's modules declare. + (lean_root / "Scratch.lean").write_text( + "theorem Demo.twin : True := trivial\n\ntheorem Demo.made : True := trivial\n", encoding="utf-8" + ) + _stub_probe( + monkeypatch, + [ + *_STUB_RECORDS, + _payload("_private.Demo.Extra.0.Demo.secret", module="Demo.Extra", type_uses=["Demo.base"]), + _payload("Demo.twin", type_uses=["Demo.base"]), + _payload("Demo.made", module="Demo.Made", type_uses=["Demo.base"]), + _payload("Demo.lost", module="Demo.Made", type_uses=["Demo.base"]), + ], + ) + + report = revision_impact(project, "chapter/base", lean_root=lean_root) + + # A private name is looked up by its source name. An indexed declaration + # in another file than the module's own gives only that file, without a + # line; a module without a source file falls back to the index alone. + assert [(helper.name, helper.path, helper.line) for helper in report.helpers] == [ + ("Demo.base_eq", "Demo.lean", 5), + ("Demo.lost", None, None), + ("Demo.made", "Scratch.lean", 3), + ("Demo.twin", "Demo.lean", None), + ("_private.Demo.Extra.0.Demo.secret", "Demo/Extra.lean", 3), + ] + + +def _rewrite_base(project: Path) -> None: + metadata = [f"article_id: {_BASE_ID}", "declaration: definition", "statement: formalized", "lean: Demo.base"] + _write_article(project, "base.md", metadata=metadata, title="Base, revised") + + +def _delete_empty(project: Path) -> None: + (project / "blueprint/roadmap/chapter/empty.md").unlink() + + +@pytest.mark.parametrize(("selector", "change"), [("chapter/base", _rewrite_base), ("chapter/empty", _delete_empty)]) +def test_revision_impact_refuses_a_roadmap_that_changes_while_it_is_read( + tmp_path: Path, monkeypatch, selector: str, change +) -> None: + project = _blueprint_project(tmp_path, "Demo") + lean_root = _stub_lean_root(tmp_path) + calls = _stub_probe(monkeypatch) + + def changed(target: Path): + change(project) + return load_runtime_graph(target) + + monkeypatch.setattr("autoform_cli.impact.load_runtime_graph", changed) + + with pytest.raises(ImpactError, match="^the roadmap changed while it was read; rerun the command$"): + revision_impact(project, selector, lean_root=lean_root) + assert calls == [] + + # --------------------------------------------------------------------------- # # The real probe # --------------------------------------------------------------------------- # From 36ab265e47b0d2c90e1388bef7416012b9342232 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 07:17:51 -0400 Subject: [PATCH 09/38] Audit open statements in project CI When roadmap/README.md sets open_statements: allowed, the verify workflow asks `autoform work assumptions` for the assumption contract and renders an open-statement probe instead of the strict one. The probe accepts sorry only as the direct proof of a theorem that an open article's lean: names, and accepts any other declaration that reaches such a theorem only when its article's Markdown dependencies reach it. It rejects sorry in a type, a helper, a where clause or a definition, sorry inherited from outside the root package, contract names that the build does not declare, and an open statement whose article records it as proved. Every article declaration is logged as an open statement, conditional, or sorry-free. The article table is embedded as one JSON string and parsed with Lean.Json, because a term with thousands of tuple entries exceeds the elaborator's and code generator's recursion limits. Without the opt-in, the helper renders the same strict probe as before. The real-Lean CI job now also runs tests/test_lake_artifact_audit.py, so the new probe scenarios and the existing Lake artifact checks run against the fixture toolchain. --- .github/workflows/tests.yml | 4 +- .../templates/github/autoform_audit.py | 431 +++++++++++++++++- .../github/workflows/autoform-verify.yml | 18 +- .../.github/autoform_audit.py | 431 +++++++++++++++++- .../.github/workflows/autoform-verify.yml | 18 +- tests/test_lake_artifact_audit.py | 421 +++++++++++++++++ 6 files changed, 1303 insertions(+), 20 deletions(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index f79dfb32..b0f055c7 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -55,8 +55,8 @@ jobs: timeout --signal=TERM --kill-after=30s 10m \ elan toolchain install "$(tr -d '\r\n' < tests/fixtures/skeleton-project/lean-toolchain)" command -v lake - - name: Run the skeleton suite against real Lean - run: timeout --signal=TERM --kill-after=30s 25m uv run pytest -q tests/test_skeleton.py + - name: Run the real-Lean suites + run: timeout --signal=TERM --kill-after=30s 25m uv run pytest -q tests/test_skeleton.py tests/test_lake_artifact_audit.py env: AUTOFORM_REQUIRE_REAL_LEAN_TESTS: "1" diff --git a/autoform_cli/templates/github/autoform_audit.py b/autoform_cli/templates/github/autoform_audit.py index b147b3e5..0f256c3d 100755 --- a/autoform_cli/templates/github/autoform_audit.py +++ b/autoform_cli/templates/github/autoform_audit.py @@ -7,10 +7,17 @@ import re import sys import tarfile +import unicodedata from pathlib import Path, PurePosixPath +from typing import NamedTuple _MAX_ILEAN_BYTES = 16 * 1024 * 1024 _TOP_LEVEL_NAME = re.compile(r'^name\s*=\s*("(?:[^"\\]|\\.)*")\s*(?:#.*)?$') +_ASSUMPTIONS_SCHEMA = "autoform-assumptions/v1" +_DECLARATION_NAME_PART = r"«([^»]+)»|([^.\s«»]+)" +_DECLARATION_NAME = re.compile( + rf"(?:{_DECLARATION_NAME_PART})(?:\.(?:{_DECLARATION_NAME_PART}))*" +) class AuditInputError(ValueError): @@ -50,6 +57,47 @@ def root_package_from_config(config: Path) -> str: return names[0] +def blueprint_policy(blueprint: Path) -> str: + """Return ``allowed`` or ``forbidden``: may theorems keep a ``sorry`` proof? + + Only ``roadmap/README.md`` sets the policy, in its frontmatter. This reads + that one key the way the CLI's frontmatter parser does. A reading that + differs from the CLI fails safe: ``forbidden`` runs the strict audit, and a + wrong ``allowed`` meets an assumption contract that refuses it. + """ + + path = blueprint / "roadmap" / "README.md" + try: + lines = path.read_bytes().decode("utf-8").splitlines() + except FileNotFoundError: + return "forbidden" + except (OSError, UnicodeError) as exc: + raise AuditInputError(f"cannot read {path}: {exc}") from exc + if not lines or lines[0].strip() != "---": + return "forbidden" + end = next((index for index in range(1, len(lines)) if lines[index].strip() == "---"), None) + if end is None: + raise AuditInputError(f"{path}: unterminated frontmatter") + policy: str | None = None + for line in lines[1:end]: + stripped = line.strip() + if not stripped or stripped.startswith("#") or ":" not in stripped: + continue + key, value = (part.strip() for part in stripped.split(":", 1)) + if key != "open_statements": + continue + if policy is not None: + raise AuditInputError(f"{path}: duplicate frontmatter key 'open_statements'") + if len(value) >= 2 and value[0] == value[-1] and value[0] in {'"', "'"}: + value = value[1:-1] + if not value: + raise AuditInputError(f"{path}: empty frontmatter value for 'open_statements'") + policy = value.casefold() + if policy not in {"allowed", "forbidden"}: + raise AuditInputError(f"{path}: 'open_statements' accepts allowed or forbidden") + return policy or "forbidden" + + def modules_from_archive(archive: Path, root_package: str) -> tuple[str, ...]: """Return modules proven to be built as part of *root_package*.""" @@ -256,6 +304,350 @@ def render_probe(modules: tuple[str, ...]) -> str: """ +class ArticleDeclaration(NamedTuple): + """One ``lean:`` declaration of a stated article, as the contract records it.""" + + name: str + article: str + is_open: bool + allowed: tuple[str, ...] + + +def load_assumption_contract(path: Path) -> tuple[ArticleDeclaration, ...]: + """Read ``autoform work assumptions --json`` output and refuse anything unexpected. + + The Markdown is the only authority on which statements are open: an open + article's theorems may keep a ``sorry`` proof, and each article's + declarations may rest only on the open statements its Markdown + dependencies reach. + """ + + try: + contract = json.loads( + path.read_bytes().decode("utf-8"), + object_pairs_hook=_unique_keys, + parse_constant=_reject_json_constant, + ) + except (OSError, ValueError) as exc: + raise AuditInputError(f"cannot read the assumption contract: {exc}") from exc + if not isinstance(contract, dict) or contract.get("schema") != _ASSUMPTIONS_SCHEMA: + raise AuditInputError(f"the assumption contract is not an {_ASSUMPTIONS_SCHEMA} object") + if contract.get("open_statements") is not True: + raise AuditInputError( + "the assumption contract says roadmap/README.md does not allow open statements; " + "refusing to accept any sorry" + ) + articles = contract.get("articles") + if not isinstance(articles, list): + raise AuditInputError("the assumption contract has no articles list") + entries: list[ArticleDeclaration] = [] + article_ids: set[str] = set() + open_owners: dict[str, str] = {} + proved_owners: dict[str, str] = {} + for article in articles: + if not isinstance(article, dict): + raise AuditInputError("the assumption contract lists an article that is not an object") + article_id = article.get("id") + if not isinstance(article_id, str) or not article_id or not _printable(article_id): + raise AuditInputError(f"the assumption contract lists an invalid article id: {article_id!r}") + if article_id in article_ids: + raise AuditInputError(f"the assumption contract lists {article_id} twice") + article_ids.add(article_id) + is_open = article.get("open") + declarations = article.get("declarations") + allowed = article.get("allowed_open_declarations") + assumes = article.get("assumes") + if ( + not isinstance(is_open, bool) + or not isinstance(declarations, list) + or not declarations + or not isinstance(allowed, list) + or not isinstance(assumes, list) + or not all(isinstance(assumed, str) for assumed in assumes) + or not isinstance(article.get("state"), str) + or not isinstance(article.get("article_id"), (str, type(None))) + ): + raise AuditInputError(f"the assumption contract has a malformed entry for {article_id}") + for name in (*declarations, *allowed): + _declaration_name_parts(name, article_id) + owners = open_owners if is_open else proved_owners + for name in declarations: + owners.setdefault(name, article_id) + allowed_names = tuple(dict.fromkeys(allowed)) + for name in dict.fromkeys(declarations): + entries.append(ArticleDeclaration(name, article_id, is_open, allowed_names)) + for name, owner in open_owners.items(): + if name in proved_owners: + raise AuditInputError( + f"{name} is an open statement of {owner} but {proved_owners[name]} records it as proved" + ) + for entry in entries: + for name in entry.allowed: + if name not in open_owners: + raise AuditInputError( + f"the assumption contract lets {entry.article} rest on {name}, " + "which no open article declares" + ) + return tuple(entries) + + +def _unique_keys(pairs: list[tuple[str, object]]) -> dict[str, object]: + result: dict[str, object] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate key {key!r}") + result[key] = value + return result + + +def _reject_json_constant(value: str) -> object: + raise ValueError(f"unsupported JSON constant {value}") + + +def _printable(text: str) -> bool: + """Reject control characters and lone surrogates before they reach Lean source.""" + + return all(unicodedata.category(character) not in {"Cc", "Cs"} for character in text) + + +def _declaration_name_parts(name: object, article: str) -> tuple[tuple[str, bool], ...]: + """Parse a declaration name as the CLI does: guillemets quote one component.""" + + if not isinstance(name, str) or not _DECLARATION_NAME.fullmatch(name) or not _printable(name): + raise AuditInputError(f"{article} names an invalid Lean declaration: {name!r}") + return tuple( + (quoted, True) if quoted else (plain, False) + for quoted, plain in re.findall(_DECLARATION_NAME_PART, name) + ) + + +def render_open_probe( + modules: tuple[str, ...], declarations: tuple[ArticleDeclaration, ...] +) -> str: + """Render the Lean program that audits *modules* against the assumption contract. + + The strict audit's rules hold with one exception: a theorem that an open + article names may keep ``sorry`` in its proof. Whatever reaches such a + statement is reported as conditional on it, and must belong to an article + whose Markdown dependencies reach it. + """ + + if not modules: + raise AuditInputError("refusing to render an empty open-statement audit") + imports = "\n".join(f"import {module}" for module in modules) + target_modules = ", ".join(_lean_name(module) for module in modules) + # One string literal, not a Lean term: a term with thousands of entries + # exceeds the elaborator's and code generator's recursion limits. + table = json.dumps( + [ + [ + _json_declaration_name(entry.name), + entry.article, + entry.is_open, + [_json_declaration_name(name) for name in entry.allowed], + ] + for entry in declarations + ], + ensure_ascii=False, + separators=(",", ":"), + ) + articles = json.dumps(table, ensure_ascii=False) + return f"""{imports} +import Lean.Util.CollectAxioms +import Lean.Elab.Command +import Lean.Data.Json + +open Lean Elab Command + +namespace AutoformOpenStatementAudit + +/-- A name spelled as its components: strings, and numbers for numeric ones. -/ +def nameOf (json : Json) : Except String Name := do + let mut name := Name.anonymous + for part in (← json.getArr?) do + match part with + | .str text => name := Name.str name text + | _ => name := Name.num name (← part.getNat?) + return name + +/-- Per article declaration: its name, its article, whether the article is an +open statement, and the open statements the declaration may rest on. -/ +def readArticles (text : String) : Except String (Array (Name × String × Bool × Array Name)) := do + (← (← Json.parse text).getArr?).mapM fun entry => do + let declName ← nameOf (← entry.getArrVal? 0) + let article ← (← entry.getArrVal? 1).getStr? + let isOpen ← (← entry.getArrVal? 2).getBool? + let allowedOpen ← (← (← entry.getArrVal? 3).getArr?).mapM nameOf + return (declName, article, isOpen, allowedOpen) + +/-- The types and constructors of one mutual inductive block refer to each +other. Nothing else in a safe environment does, so the walk treats a block as +one node and needs no cycle handling. -/ +def blockOf (env : Environment) (declName : Name) : Name := + let induct := match env.find? declName with + | some (.ctorInfo info) => info.induct + | _ => declName + match env.find? induct with + | some (.inductInfo info) => info.all.head?.getD induct + | _ => declName + +/-- The constants a node refers to, as `collectAxioms` follows them, except that +an open statement's proof is not entered: a dependent rests on the statement, +whatever its proof uses. -/ +def edges (env : Environment) (openSet : Std.HashSet Name) (node : Name) : Array Name := + match env.find? node with + | some (.inductInfo info) => + info.all.foldl (init := (#[] : Array Name)) fun used induct => + match env.find? induct with + | some (.inductInfo block) => + block.ctors.foldl (init := used ++ block.type.getUsedConstants) fun used ctor => + used ++ ((env.find? ctor).map (·.type.getUsedConstants)).getD #[] + | _ => used + | some (.thmInfo info) => + if openSet.contains node then info.type.getUsedConstants + else info.type.getUsedConstants ++ info.value.getUsedConstants + | some (.defnInfo info) => info.type.getUsedConstants ++ info.value.getUsedConstants + | some (.opaqueInfo info) => info.type.getUsedConstants ++ info.value.getUsedConstants + | some info => info.type.getUsedConstants + | none => #[] + +/-- The open statements a node rests on. A constant outside the root package +contributes none: the audit requires those to be free of `sorry`. -/ +partial def openHits (env : Environment) (isRoot : Name → Bool) (openSet : Std.HashSet Name) + (node : Name) : StateM (Std.HashMap Name (Array Name)) (Array Name) := do + if let some known := (← get).get? node then + return known + -- In progress. Only unsafe recursion comes back here, and the audit rejects it. + modify (·.insert node #[]) + let mut hits : Array Name := if openSet.contains node then #[node] else #[] + for used in edges env openSet node do + if used == ``sorryAx || !isRoot used then + continue + let target := blockOf env used + if target == node then + continue + for hit in (← openHits env isRoot openSet target) do + unless hits.contains hit do + hits := hits.push hit + hits := hits.qsort Name.lt + modify (·.insert node hits) + return hits + +def nameList (names : Array Name) : MessageData := + MessageData.joinSep (names.toList.map MessageData.ofName) ", " + +end AutoformOpenStatementAudit + +run_cmd do + let targetModules : List Name := [{target_modules}] + let allowed : List Name := [``propext, ``Classical.choice, ``Quot.sound] + let articles ← match AutoformOpenStatementAudit.readArticles {articles} with + | .ok articles => pure articles + | .error message => throwError "cannot read the article table: {{message}}" + let env ← getEnv + let isRoot (declName : Name) : Bool := + match env.getModuleIdxFor? declName with + | some moduleIdx => targetModules.contains env.header.moduleNames[moduleIdx.toNat]! + | none => false + let mut openSet : Std.HashSet Name := {{}} + for (declName, _, isOpen, _) in articles do + if isOpen && isRoot declName then + if let some (.thmInfo _) := env.find? declName then + openSet := openSet.insert declName + let mut roots : Array Name := #[] + for (declName, _) in env.constants do + if isRoot declName then + roots := roots.push declName + roots := roots.qsort Name.lt + let mut errors : Array MessageData := #[] + let mut hitCache : Std.HashMap Name (Array Name) := {{}} + let mut externalSorry : Std.HashMap Name Bool := {{}} + for declName in roots do + let some info := env.find? declName | continue + let reported := errors.size + if info.isUnsafe || info.isPartial then + errors := errors.push m!"unsafe or partial declaration: {{declName}}" + let usedAxioms ← Lean.collectAxioms declName + for usedAxiom in usedAxioms do + unless usedAxiom == ``sorryAx || allowed.contains usedAxiom do + errors := errors.push m!"{{declName}} depends on unexpected axiom {{usedAxiom}}" + let typeConstants := info.type.getUsedConstants + -- Values are read by kind: `ConstantInfo.value?` hides theorem proofs by + -- default, and a hidden proof would pass this check vacuously. + let valueConstants := match info with + | .thmInfo val => val.value.getUsedConstants + | .defnInfo val => val.value.getUsedConstants + | .opaqueInfo val => val.value.getUsedConstants + | _ => #[] + if typeConstants.contains ``sorryAx then + errors := errors.push m!"{{declName}} has sorry in its statement; state the claim in full and keep sorry only in the proof" + else if valueConstants.contains ``sorryAx && !openSet.contains declName then + errors := errors.push m!"{{declName}} contains sorry but is not an open statement: only a theorem that an open article's lean: names may keep a sorry, written directly in its own proof, not in a helper, where clause or definition" + let mut external : Array Name := #[] + for used in typeConstants ++ valueConstants do + if used == ``sorryAx || isRoot used || external.contains used then + continue + external := external.push used + let mut tainted := false + if let some known := externalSorry.get? used then + tainted := known + else + tainted := (← Lean.collectAxioms used).contains ``sorryAx + externalSorry := externalSorry.insert used tainted + if tainted then + errors := errors.push m!"{{declName}} uses {{used}}, which is outside the root package and depends on sorry" + let (hits, cache) := Id.run ((AutoformOpenStatementAudit.openHits env isRoot openSet + (AutoformOpenStatementAudit.blockOf env declName)).run hitCache) + hitCache := cache + if errors.size == reported && usedAxioms.contains ``sorryAx && hits.isEmpty then + errors := errors.push m!"{{declName}} depends on sorry outside every declared open statement" + let mut openCount : Nat := 0 + let mut conditionalCount : Nat := 0 + for (declName, article, isOpen, allowedOpen) in articles do + match env.find? declName with + | none => + errors := errors.push m!"{{declName}} [{{article}}] is not a declaration of the Lean build; fix the article's lean: name or build the module that declares it" + | some info => + if !isRoot declName then + if (← Lean.collectAxioms declName).contains ``sorryAx then + errors := errors.push m!"{{declName}} [{{article}}] is outside the root package and depends on sorry" + else + logInfo m!"sorry-free: {{declName}} [{{article}}]" + else + let (hits, cache) := Id.run ((AutoformOpenStatementAudit.openHits env isRoot openSet + (AutoformOpenStatementAudit.blockOf env declName)).run hitCache) + hitCache := cache + let undeclared := hits.filter (fun hit => !allowedOpen.contains hit) + unless undeclared.isEmpty do + errors := errors.push m!"{{declName}} [{{article}}] rests on open statement(s) {{AutoformOpenStatementAudit.nameList undeclared}}, which its article's Markdown dependencies do not reach; add the dependency to the article or stop using them" + if openSet.contains declName then + unless isOpen do + errors := errors.push m!"{{declName}} [{{article}}] is an open statement, but its article records it as proved" + openCount := openCount + 1 + let ownSorry := match info with + | .thmInfo val => val.value.getUsedConstants.contains ``sorryAx + | _ => false + if ownSorry then + logInfo m!"open statement (proof is sorry): {{declName}} [{{article}}]" + else if (← Lean.collectAxioms declName).contains ``sorryAx then + logInfo m!"open statement (proof depends on sorry elsewhere): {{declName}} [{{article}}]" + else + logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" + else if !hits.isEmpty then + conditionalCount := conditionalCount + 1 + logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{AutoformOpenStatementAudit.nameList hits}}" + else unless (← Lean.collectAxioms declName).contains ``sorryAx do + logInfo m!"sorry-free: {{declName}} [{{article}}]" + for error in errors do + logError error + if roots.isEmpty then + throwError "open-statement audit found no root-package declarations" + unless errors.isEmpty do + throwError "root-package declarations failed the open-statement audit" + logInfo m!"kernel trust clean except declared open statements ({{roots.size}} root-package declaration(s) audited; {{openCount}} open statement(s), {{conditionalCount}} conditional declaration(s))" +""" + + def _lean_name(module: str) -> str: result = "Name.anonymous" for part in _module_parts(module, module): @@ -263,19 +655,35 @@ def _lean_name(module: str) -> str: return result +def _json_declaration_name(name: str) -> list[str | int]: + """Spell a declaration name as its components, as the skeleton probe does.""" + + return [ + int(part) if not quoted and part.isascii() and part.isdigit() else part + for part, quoted in _declaration_name_parts(name, name) + ] + + def main(argv: list[str] | None = None) -> int: arguments = sys.argv[1:] if argv is None else argv - if len(arguments) == 2 and arguments[0] == "--root-package": + if len(arguments) == 2 and arguments[0] in {"--root-package", "--policy"}: + read = root_package_from_config if arguments[0] == "--root-package" else blueprint_policy try: - print(root_package_from_config(Path(arguments[1]))) + print(read(Path(arguments[1]))) except AuditInputError as exc: print(f"error: {exc}", file=sys.stderr) return 1 return 0 - if len(arguments) != 3: + contract: Path | None = None + if len(arguments) == 5 and arguments[0] == "--open-statements": + contract = Path(arguments[1]) + arguments = arguments[2:] + if len(arguments) != 3 or arguments[0] in {"--policy", "--open-statements"}: print( "usage: autoform_audit.py --root-package EVALUATED_CONFIG\n" - " or: autoform_audit.py ROOT_PACKAGE ROOT_BUILD_ARCHIVE OUTPUT_PROBE", + " or: autoform_audit.py --policy BLUEPRINT_DIR\n" + " or: autoform_audit.py ROOT_PACKAGE ROOT_BUILD_ARCHIVE OUTPUT_PROBE\n" + " or: autoform_audit.py --open-statements CONTRACT ROOT_PACKAGE ROOT_BUILD_ARCHIVE OUTPUT_PROBE", file=sys.stderr, ) return 2 @@ -283,12 +691,23 @@ def main(argv: list[str] | None = None) -> int: archive, output = map(Path, arguments[1:]) try: modules = modules_from_archive(archive, root_package) - probe = render_probe(modules) + if contract is None: + probe = render_probe(modules) + else: + declarations = load_assumption_contract(contract) + probe = render_open_probe(modules, declarations) output.write_text(probe, encoding="utf-8") except (AuditInputError, OSError) as exc: print(f"error: {exc}", file=sys.stderr) return 1 - print(f"prepared kernel-trust audit for {len(modules)} root-package module(s)") + if contract is None: + print(f"prepared kernel-trust audit for {len(modules)} root-package module(s)") + else: + candidates = len({entry.name for entry in declarations if entry.is_open}) + print( + f"prepared open-statement audit for {len(modules)} root-package module(s) " + f"and {candidates} open statement candidate(s)" + ) return 0 diff --git a/autoform_cli/templates/github/workflows/autoform-verify.yml b/autoform_cli/templates/github/workflows/autoform-verify.yml index 8880525b..427171b0 100644 --- a/autoform_cli/templates/github/workflows/autoform-verify.yml +++ b/autoform_cli/templates/github/workflows/autoform-verify.yml @@ -91,11 +91,23 @@ jobs: set -euo pipefail archive="$(mktemp "${RUNNER_TEMP:-/tmp}/autoform-root-build.XXXXXX.tgz")" probe="$(mktemp "${RUNNER_TEMP:-/tmp}/autoform-axiom-probe.XXXXXX.lean")" - trap 'rm -f "$archive" "$probe"' EXIT + contract="$(mktemp "${RUNNER_TEMP:-/tmp}/autoform-assumptions.XXXXXX.json")" + trap 'rm -f "$archive" "$probe" "$contract"' EXIT # Lake resolves both manifest languages and package/custom buildDir. # `lake pack` archives only the root package's actual build directory, # leaving dependency package artifacts outside the audit boundary. lake pack "$archive" - python3 .github/autoform_audit.py \ - "$AUTOFORM_ROOT_PACKAGE" "$archive" "$probe" + # roadmap/README.md may opt in to open statements (theorems whose proof + # is still sorry). Only then does the audit accept sorry, and only in + # the statements the Markdown declares open. + policy="$(python3 .github/autoform_audit.py --policy blueprint)" + if [ "$policy" = "allowed" ]; then + uvx --from "git+${AUTOFORM_SOURCE}@${AUTOFORM_REF}" \ + autoform work assumptions blueprint --json > "$contract" + python3 .github/autoform_audit.py --open-statements "$contract" \ + "$AUTOFORM_ROOT_PACKAGE" "$archive" "$probe" + else + python3 .github/autoform_audit.py \ + "$AUTOFORM_ROOT_PACKAGE" "$archive" "$probe" + fi lake env lean "$probe" diff --git a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py index b147b3e5..0f256c3d 100755 --- a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py +++ b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py @@ -7,10 +7,17 @@ import re import sys import tarfile +import unicodedata from pathlib import Path, PurePosixPath +from typing import NamedTuple _MAX_ILEAN_BYTES = 16 * 1024 * 1024 _TOP_LEVEL_NAME = re.compile(r'^name\s*=\s*("(?:[^"\\]|\\.)*")\s*(?:#.*)?$') +_ASSUMPTIONS_SCHEMA = "autoform-assumptions/v1" +_DECLARATION_NAME_PART = r"«([^»]+)»|([^.\s«»]+)" +_DECLARATION_NAME = re.compile( + rf"(?:{_DECLARATION_NAME_PART})(?:\.(?:{_DECLARATION_NAME_PART}))*" +) class AuditInputError(ValueError): @@ -50,6 +57,47 @@ def root_package_from_config(config: Path) -> str: return names[0] +def blueprint_policy(blueprint: Path) -> str: + """Return ``allowed`` or ``forbidden``: may theorems keep a ``sorry`` proof? + + Only ``roadmap/README.md`` sets the policy, in its frontmatter. This reads + that one key the way the CLI's frontmatter parser does. A reading that + differs from the CLI fails safe: ``forbidden`` runs the strict audit, and a + wrong ``allowed`` meets an assumption contract that refuses it. + """ + + path = blueprint / "roadmap" / "README.md" + try: + lines = path.read_bytes().decode("utf-8").splitlines() + except FileNotFoundError: + return "forbidden" + except (OSError, UnicodeError) as exc: + raise AuditInputError(f"cannot read {path}: {exc}") from exc + if not lines or lines[0].strip() != "---": + return "forbidden" + end = next((index for index in range(1, len(lines)) if lines[index].strip() == "---"), None) + if end is None: + raise AuditInputError(f"{path}: unterminated frontmatter") + policy: str | None = None + for line in lines[1:end]: + stripped = line.strip() + if not stripped or stripped.startswith("#") or ":" not in stripped: + continue + key, value = (part.strip() for part in stripped.split(":", 1)) + if key != "open_statements": + continue + if policy is not None: + raise AuditInputError(f"{path}: duplicate frontmatter key 'open_statements'") + if len(value) >= 2 and value[0] == value[-1] and value[0] in {'"', "'"}: + value = value[1:-1] + if not value: + raise AuditInputError(f"{path}: empty frontmatter value for 'open_statements'") + policy = value.casefold() + if policy not in {"allowed", "forbidden"}: + raise AuditInputError(f"{path}: 'open_statements' accepts allowed or forbidden") + return policy or "forbidden" + + def modules_from_archive(archive: Path, root_package: str) -> tuple[str, ...]: """Return modules proven to be built as part of *root_package*.""" @@ -256,6 +304,350 @@ def render_probe(modules: tuple[str, ...]) -> str: """ +class ArticleDeclaration(NamedTuple): + """One ``lean:`` declaration of a stated article, as the contract records it.""" + + name: str + article: str + is_open: bool + allowed: tuple[str, ...] + + +def load_assumption_contract(path: Path) -> tuple[ArticleDeclaration, ...]: + """Read ``autoform work assumptions --json`` output and refuse anything unexpected. + + The Markdown is the only authority on which statements are open: an open + article's theorems may keep a ``sorry`` proof, and each article's + declarations may rest only on the open statements its Markdown + dependencies reach. + """ + + try: + contract = json.loads( + path.read_bytes().decode("utf-8"), + object_pairs_hook=_unique_keys, + parse_constant=_reject_json_constant, + ) + except (OSError, ValueError) as exc: + raise AuditInputError(f"cannot read the assumption contract: {exc}") from exc + if not isinstance(contract, dict) or contract.get("schema") != _ASSUMPTIONS_SCHEMA: + raise AuditInputError(f"the assumption contract is not an {_ASSUMPTIONS_SCHEMA} object") + if contract.get("open_statements") is not True: + raise AuditInputError( + "the assumption contract says roadmap/README.md does not allow open statements; " + "refusing to accept any sorry" + ) + articles = contract.get("articles") + if not isinstance(articles, list): + raise AuditInputError("the assumption contract has no articles list") + entries: list[ArticleDeclaration] = [] + article_ids: set[str] = set() + open_owners: dict[str, str] = {} + proved_owners: dict[str, str] = {} + for article in articles: + if not isinstance(article, dict): + raise AuditInputError("the assumption contract lists an article that is not an object") + article_id = article.get("id") + if not isinstance(article_id, str) or not article_id or not _printable(article_id): + raise AuditInputError(f"the assumption contract lists an invalid article id: {article_id!r}") + if article_id in article_ids: + raise AuditInputError(f"the assumption contract lists {article_id} twice") + article_ids.add(article_id) + is_open = article.get("open") + declarations = article.get("declarations") + allowed = article.get("allowed_open_declarations") + assumes = article.get("assumes") + if ( + not isinstance(is_open, bool) + or not isinstance(declarations, list) + or not declarations + or not isinstance(allowed, list) + or not isinstance(assumes, list) + or not all(isinstance(assumed, str) for assumed in assumes) + or not isinstance(article.get("state"), str) + or not isinstance(article.get("article_id"), (str, type(None))) + ): + raise AuditInputError(f"the assumption contract has a malformed entry for {article_id}") + for name in (*declarations, *allowed): + _declaration_name_parts(name, article_id) + owners = open_owners if is_open else proved_owners + for name in declarations: + owners.setdefault(name, article_id) + allowed_names = tuple(dict.fromkeys(allowed)) + for name in dict.fromkeys(declarations): + entries.append(ArticleDeclaration(name, article_id, is_open, allowed_names)) + for name, owner in open_owners.items(): + if name in proved_owners: + raise AuditInputError( + f"{name} is an open statement of {owner} but {proved_owners[name]} records it as proved" + ) + for entry in entries: + for name in entry.allowed: + if name not in open_owners: + raise AuditInputError( + f"the assumption contract lets {entry.article} rest on {name}, " + "which no open article declares" + ) + return tuple(entries) + + +def _unique_keys(pairs: list[tuple[str, object]]) -> dict[str, object]: + result: dict[str, object] = {} + for key, value in pairs: + if key in result: + raise ValueError(f"duplicate key {key!r}") + result[key] = value + return result + + +def _reject_json_constant(value: str) -> object: + raise ValueError(f"unsupported JSON constant {value}") + + +def _printable(text: str) -> bool: + """Reject control characters and lone surrogates before they reach Lean source.""" + + return all(unicodedata.category(character) not in {"Cc", "Cs"} for character in text) + + +def _declaration_name_parts(name: object, article: str) -> tuple[tuple[str, bool], ...]: + """Parse a declaration name as the CLI does: guillemets quote one component.""" + + if not isinstance(name, str) or not _DECLARATION_NAME.fullmatch(name) or not _printable(name): + raise AuditInputError(f"{article} names an invalid Lean declaration: {name!r}") + return tuple( + (quoted, True) if quoted else (plain, False) + for quoted, plain in re.findall(_DECLARATION_NAME_PART, name) + ) + + +def render_open_probe( + modules: tuple[str, ...], declarations: tuple[ArticleDeclaration, ...] +) -> str: + """Render the Lean program that audits *modules* against the assumption contract. + + The strict audit's rules hold with one exception: a theorem that an open + article names may keep ``sorry`` in its proof. Whatever reaches such a + statement is reported as conditional on it, and must belong to an article + whose Markdown dependencies reach it. + """ + + if not modules: + raise AuditInputError("refusing to render an empty open-statement audit") + imports = "\n".join(f"import {module}" for module in modules) + target_modules = ", ".join(_lean_name(module) for module in modules) + # One string literal, not a Lean term: a term with thousands of entries + # exceeds the elaborator's and code generator's recursion limits. + table = json.dumps( + [ + [ + _json_declaration_name(entry.name), + entry.article, + entry.is_open, + [_json_declaration_name(name) for name in entry.allowed], + ] + for entry in declarations + ], + ensure_ascii=False, + separators=(",", ":"), + ) + articles = json.dumps(table, ensure_ascii=False) + return f"""{imports} +import Lean.Util.CollectAxioms +import Lean.Elab.Command +import Lean.Data.Json + +open Lean Elab Command + +namespace AutoformOpenStatementAudit + +/-- A name spelled as its components: strings, and numbers for numeric ones. -/ +def nameOf (json : Json) : Except String Name := do + let mut name := Name.anonymous + for part in (← json.getArr?) do + match part with + | .str text => name := Name.str name text + | _ => name := Name.num name (← part.getNat?) + return name + +/-- Per article declaration: its name, its article, whether the article is an +open statement, and the open statements the declaration may rest on. -/ +def readArticles (text : String) : Except String (Array (Name × String × Bool × Array Name)) := do + (← (← Json.parse text).getArr?).mapM fun entry => do + let declName ← nameOf (← entry.getArrVal? 0) + let article ← (← entry.getArrVal? 1).getStr? + let isOpen ← (← entry.getArrVal? 2).getBool? + let allowedOpen ← (← (← entry.getArrVal? 3).getArr?).mapM nameOf + return (declName, article, isOpen, allowedOpen) + +/-- The types and constructors of one mutual inductive block refer to each +other. Nothing else in a safe environment does, so the walk treats a block as +one node and needs no cycle handling. -/ +def blockOf (env : Environment) (declName : Name) : Name := + let induct := match env.find? declName with + | some (.ctorInfo info) => info.induct + | _ => declName + match env.find? induct with + | some (.inductInfo info) => info.all.head?.getD induct + | _ => declName + +/-- The constants a node refers to, as `collectAxioms` follows them, except that +an open statement's proof is not entered: a dependent rests on the statement, +whatever its proof uses. -/ +def edges (env : Environment) (openSet : Std.HashSet Name) (node : Name) : Array Name := + match env.find? node with + | some (.inductInfo info) => + info.all.foldl (init := (#[] : Array Name)) fun used induct => + match env.find? induct with + | some (.inductInfo block) => + block.ctors.foldl (init := used ++ block.type.getUsedConstants) fun used ctor => + used ++ ((env.find? ctor).map (·.type.getUsedConstants)).getD #[] + | _ => used + | some (.thmInfo info) => + if openSet.contains node then info.type.getUsedConstants + else info.type.getUsedConstants ++ info.value.getUsedConstants + | some (.defnInfo info) => info.type.getUsedConstants ++ info.value.getUsedConstants + | some (.opaqueInfo info) => info.type.getUsedConstants ++ info.value.getUsedConstants + | some info => info.type.getUsedConstants + | none => #[] + +/-- The open statements a node rests on. A constant outside the root package +contributes none: the audit requires those to be free of `sorry`. -/ +partial def openHits (env : Environment) (isRoot : Name → Bool) (openSet : Std.HashSet Name) + (node : Name) : StateM (Std.HashMap Name (Array Name)) (Array Name) := do + if let some known := (← get).get? node then + return known + -- In progress. Only unsafe recursion comes back here, and the audit rejects it. + modify (·.insert node #[]) + let mut hits : Array Name := if openSet.contains node then #[node] else #[] + for used in edges env openSet node do + if used == ``sorryAx || !isRoot used then + continue + let target := blockOf env used + if target == node then + continue + for hit in (← openHits env isRoot openSet target) do + unless hits.contains hit do + hits := hits.push hit + hits := hits.qsort Name.lt + modify (·.insert node hits) + return hits + +def nameList (names : Array Name) : MessageData := + MessageData.joinSep (names.toList.map MessageData.ofName) ", " + +end AutoformOpenStatementAudit + +run_cmd do + let targetModules : List Name := [{target_modules}] + let allowed : List Name := [``propext, ``Classical.choice, ``Quot.sound] + let articles ← match AutoformOpenStatementAudit.readArticles {articles} with + | .ok articles => pure articles + | .error message => throwError "cannot read the article table: {{message}}" + let env ← getEnv + let isRoot (declName : Name) : Bool := + match env.getModuleIdxFor? declName with + | some moduleIdx => targetModules.contains env.header.moduleNames[moduleIdx.toNat]! + | none => false + let mut openSet : Std.HashSet Name := {{}} + for (declName, _, isOpen, _) in articles do + if isOpen && isRoot declName then + if let some (.thmInfo _) := env.find? declName then + openSet := openSet.insert declName + let mut roots : Array Name := #[] + for (declName, _) in env.constants do + if isRoot declName then + roots := roots.push declName + roots := roots.qsort Name.lt + let mut errors : Array MessageData := #[] + let mut hitCache : Std.HashMap Name (Array Name) := {{}} + let mut externalSorry : Std.HashMap Name Bool := {{}} + for declName in roots do + let some info := env.find? declName | continue + let reported := errors.size + if info.isUnsafe || info.isPartial then + errors := errors.push m!"unsafe or partial declaration: {{declName}}" + let usedAxioms ← Lean.collectAxioms declName + for usedAxiom in usedAxioms do + unless usedAxiom == ``sorryAx || allowed.contains usedAxiom do + errors := errors.push m!"{{declName}} depends on unexpected axiom {{usedAxiom}}" + let typeConstants := info.type.getUsedConstants + -- Values are read by kind: `ConstantInfo.value?` hides theorem proofs by + -- default, and a hidden proof would pass this check vacuously. + let valueConstants := match info with + | .thmInfo val => val.value.getUsedConstants + | .defnInfo val => val.value.getUsedConstants + | .opaqueInfo val => val.value.getUsedConstants + | _ => #[] + if typeConstants.contains ``sorryAx then + errors := errors.push m!"{{declName}} has sorry in its statement; state the claim in full and keep sorry only in the proof" + else if valueConstants.contains ``sorryAx && !openSet.contains declName then + errors := errors.push m!"{{declName}} contains sorry but is not an open statement: only a theorem that an open article's lean: names may keep a sorry, written directly in its own proof, not in a helper, where clause or definition" + let mut external : Array Name := #[] + for used in typeConstants ++ valueConstants do + if used == ``sorryAx || isRoot used || external.contains used then + continue + external := external.push used + let mut tainted := false + if let some known := externalSorry.get? used then + tainted := known + else + tainted := (← Lean.collectAxioms used).contains ``sorryAx + externalSorry := externalSorry.insert used tainted + if tainted then + errors := errors.push m!"{{declName}} uses {{used}}, which is outside the root package and depends on sorry" + let (hits, cache) := Id.run ((AutoformOpenStatementAudit.openHits env isRoot openSet + (AutoformOpenStatementAudit.blockOf env declName)).run hitCache) + hitCache := cache + if errors.size == reported && usedAxioms.contains ``sorryAx && hits.isEmpty then + errors := errors.push m!"{{declName}} depends on sorry outside every declared open statement" + let mut openCount : Nat := 0 + let mut conditionalCount : Nat := 0 + for (declName, article, isOpen, allowedOpen) in articles do + match env.find? declName with + | none => + errors := errors.push m!"{{declName}} [{{article}}] is not a declaration of the Lean build; fix the article's lean: name or build the module that declares it" + | some info => + if !isRoot declName then + if (← Lean.collectAxioms declName).contains ``sorryAx then + errors := errors.push m!"{{declName}} [{{article}}] is outside the root package and depends on sorry" + else + logInfo m!"sorry-free: {{declName}} [{{article}}]" + else + let (hits, cache) := Id.run ((AutoformOpenStatementAudit.openHits env isRoot openSet + (AutoformOpenStatementAudit.blockOf env declName)).run hitCache) + hitCache := cache + let undeclared := hits.filter (fun hit => !allowedOpen.contains hit) + unless undeclared.isEmpty do + errors := errors.push m!"{{declName}} [{{article}}] rests on open statement(s) {{AutoformOpenStatementAudit.nameList undeclared}}, which its article's Markdown dependencies do not reach; add the dependency to the article or stop using them" + if openSet.contains declName then + unless isOpen do + errors := errors.push m!"{{declName}} [{{article}}] is an open statement, but its article records it as proved" + openCount := openCount + 1 + let ownSorry := match info with + | .thmInfo val => val.value.getUsedConstants.contains ``sorryAx + | _ => false + if ownSorry then + logInfo m!"open statement (proof is sorry): {{declName}} [{{article}}]" + else if (← Lean.collectAxioms declName).contains ``sorryAx then + logInfo m!"open statement (proof depends on sorry elsewhere): {{declName}} [{{article}}]" + else + logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" + else if !hits.isEmpty then + conditionalCount := conditionalCount + 1 + logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{AutoformOpenStatementAudit.nameList hits}}" + else unless (← Lean.collectAxioms declName).contains ``sorryAx do + logInfo m!"sorry-free: {{declName}} [{{article}}]" + for error in errors do + logError error + if roots.isEmpty then + throwError "open-statement audit found no root-package declarations" + unless errors.isEmpty do + throwError "root-package declarations failed the open-statement audit" + logInfo m!"kernel trust clean except declared open statements ({{roots.size}} root-package declaration(s) audited; {{openCount}} open statement(s), {{conditionalCount}} conditional declaration(s))" +""" + + def _lean_name(module: str) -> str: result = "Name.anonymous" for part in _module_parts(module, module): @@ -263,19 +655,35 @@ def _lean_name(module: str) -> str: return result +def _json_declaration_name(name: str) -> list[str | int]: + """Spell a declaration name as its components, as the skeleton probe does.""" + + return [ + int(part) if not quoted and part.isascii() and part.isdigit() else part + for part, quoted in _declaration_name_parts(name, name) + ] + + def main(argv: list[str] | None = None) -> int: arguments = sys.argv[1:] if argv is None else argv - if len(arguments) == 2 and arguments[0] == "--root-package": + if len(arguments) == 2 and arguments[0] in {"--root-package", "--policy"}: + read = root_package_from_config if arguments[0] == "--root-package" else blueprint_policy try: - print(root_package_from_config(Path(arguments[1]))) + print(read(Path(arguments[1]))) except AuditInputError as exc: print(f"error: {exc}", file=sys.stderr) return 1 return 0 - if len(arguments) != 3: + contract: Path | None = None + if len(arguments) == 5 and arguments[0] == "--open-statements": + contract = Path(arguments[1]) + arguments = arguments[2:] + if len(arguments) != 3 or arguments[0] in {"--policy", "--open-statements"}: print( "usage: autoform_audit.py --root-package EVALUATED_CONFIG\n" - " or: autoform_audit.py ROOT_PACKAGE ROOT_BUILD_ARCHIVE OUTPUT_PROBE", + " or: autoform_audit.py --policy BLUEPRINT_DIR\n" + " or: autoform_audit.py ROOT_PACKAGE ROOT_BUILD_ARCHIVE OUTPUT_PROBE\n" + " or: autoform_audit.py --open-statements CONTRACT ROOT_PACKAGE ROOT_BUILD_ARCHIVE OUTPUT_PROBE", file=sys.stderr, ) return 2 @@ -283,12 +691,23 @@ def main(argv: list[str] | None = None) -> int: archive, output = map(Path, arguments[1:]) try: modules = modules_from_archive(archive, root_package) - probe = render_probe(modules) + if contract is None: + probe = render_probe(modules) + else: + declarations = load_assumption_contract(contract) + probe = render_open_probe(modules, declarations) output.write_text(probe, encoding="utf-8") except (AuditInputError, OSError) as exc: print(f"error: {exc}", file=sys.stderr) return 1 - print(f"prepared kernel-trust audit for {len(modules)} root-package module(s)") + if contract is None: + print(f"prepared kernel-trust audit for {len(modules)} root-package module(s)") + else: + candidates = len({entry.name for entry in declarations if entry.is_open}) + print( + f"prepared open-statement audit for {len(modules)} root-package module(s) " + f"and {candidates} open statement candidate(s)" + ) return 0 diff --git a/skills/setup/assets/cabannes-thesis-project/.github/workflows/autoform-verify.yml b/skills/setup/assets/cabannes-thesis-project/.github/workflows/autoform-verify.yml index 7341ac5a..bdb2b8ab 100644 --- a/skills/setup/assets/cabannes-thesis-project/.github/workflows/autoform-verify.yml +++ b/skills/setup/assets/cabannes-thesis-project/.github/workflows/autoform-verify.yml @@ -91,11 +91,23 @@ jobs: set -euo pipefail archive="$(mktemp "${RUNNER_TEMP:-/tmp}/autoform-root-build.XXXXXX.tgz")" probe="$(mktemp "${RUNNER_TEMP:-/tmp}/autoform-axiom-probe.XXXXXX.lean")" - trap 'rm -f "$archive" "$probe"' EXIT + contract="$(mktemp "${RUNNER_TEMP:-/tmp}/autoform-assumptions.XXXXXX.json")" + trap 'rm -f "$archive" "$probe" "$contract"' EXIT # Lake resolves both manifest languages and package/custom buildDir. # `lake pack` archives only the root package's actual build directory, # leaving dependency package artifacts outside the audit boundary. lake pack "$archive" - python3 .github/autoform_audit.py \ - "$AUTOFORM_ROOT_PACKAGE" "$archive" "$probe" + # roadmap/README.md may opt in to open statements (theorems whose proof + # is still sorry). Only then does the audit accept sorry, and only in + # the statements the Markdown declares open. + policy="$(python3 .github/autoform_audit.py --policy blueprint)" + if [ "$policy" = "allowed" ]; then + uvx --from "git+${AUTOFORM_SOURCE}@${AUTOFORM_REF}" \ + autoform work assumptions blueprint --json > "$contract" + python3 .github/autoform_audit.py --open-statements "$contract" \ + "$AUTOFORM_ROOT_PACKAGE" "$archive" "$probe" + else + python3 .github/autoform_audit.py \ + "$AUTOFORM_ROOT_PACKAGE" "$archive" "$probe" + fi lake env lean "$probe" diff --git a/tests/test_lake_artifact_audit.py b/tests/test_lake_artifact_audit.py index 26892f91..e1ef465c 100644 --- a/tests/test_lake_artifact_audit.py +++ b/tests/test_lake_artifact_audit.py @@ -399,3 +399,424 @@ def test_example_and_template_helpers_are_identical(repo_root: Path) -> None: example = repo_root / "skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py" assert example.read_bytes() == template.read_bytes() + + +def test_workflows_audit_open_statements_only_when_the_roadmap_allows_them(repo_root: Path) -> None: + for workflow in ( + repo_root / "autoform_cli/templates/github/workflows/autoform-verify.yml", + repo_root / "skills/setup/assets/cabannes-thesis-project/.github/workflows/autoform-verify.yml", + ): + text = workflow.read_text(encoding="utf-8") + assert 'policy="$(python3 .github/autoform_audit.py --policy blueprint)"' in text + assert 'if [ "$policy" = "allowed" ]; then' in text + assert "autoform work assumptions blueprint --json > \"$contract\"" in text + assert 'python3 .github/autoform_audit.py --open-statements "$contract"' in text + + +def _roadmap(blueprint: Path, text: str | bytes) -> None: + path = blueprint / "roadmap/README.md" + path.parent.mkdir(parents=True, exist_ok=True) + if isinstance(text, bytes): + path.write_bytes(text) + else: + path.write_text(text, encoding="utf-8") + + +@pytest.mark.parametrize( + ("text", "expected"), + [ + (None, "forbidden"), + ("# Roadmap\n", "forbidden"), + ("---\ntitle: Roadmap\n---\n# Roadmap\n", "forbidden"), + ("---\nopen_statements: allowed\n---\n", "allowed"), + ("---\nopen_statements: forbidden\n---\n", "forbidden"), + ('---\nopen_statements: "allowed"\n---\n', "allowed"), + ("---\n# policy\n\nopen_statements: 'Allowed'\n---\n", "allowed"), + ("---\nopen_statements: ALLOWED\n---\n", "allowed"), + ("# Roadmap\n---\nopen_statements: allowed\n---\n", "forbidden"), + ], +) +def test_policy_reads_the_roadmap_frontmatter( + helper: ModuleType, tmp_path: Path, text: str | None, expected: str +) -> None: + blueprint = tmp_path / "blueprint" + blueprint.mkdir() + if text is not None: + _roadmap(blueprint, text) + + assert helper.blueprint_policy(blueprint) == expected + + +def test_policy_ignores_every_page_but_the_roadmap(helper: ModuleType, tmp_path: Path) -> None: + blueprint = tmp_path / "blueprint" + _write(blueprint / "README.md", "---\nopen_statements: allowed\n---\n") + _write(blueprint / "roadmap/part/README.md", "---\nopen_statements: allowed\n---\n") + _roadmap(blueprint, "---\ntitle: Roadmap\n---\n") + + assert helper.blueprint_policy(blueprint) == "forbidden" + + +@pytest.mark.parametrize( + ("text", "message"), + [ + ("---\nopen_statements: allowed\nopen_statements: allowed\n---\n", "duplicate frontmatter key"), + ("---\nopen_statements: maybe\n---\n", "accepts allowed or forbidden"), + ("---\nopen_statements:\n---\n", "empty frontmatter value"), + ('---\nopen_statements: ""\n---\n', "empty frontmatter value"), + ("---\nopen_statements: allowed\n", "unterminated frontmatter"), + (b"---\nopen_statements: allowed\xff\n---\n", "cannot read"), + ], +) +def test_policy_refuses_ambiguous_frontmatter( + helper: ModuleType, tmp_path: Path, text: str | bytes, message: str +) -> None: + blueprint = tmp_path / "blueprint" + _roadmap(blueprint, text) + + with pytest.raises(helper.AuditInputError, match=message): + helper.blueprint_policy(blueprint) + + +def _article( + article_id: str, + declarations: list[str], + *, + is_open: bool = False, + allowed: list[str] | None = None, +) -> dict[str, object]: + return { + "allowed_open_declarations": allowed or [], + "article_id": None, + "assumes": [], + "declarations": declarations, + "id": article_id, + "open": is_open, + "state": "stated" if is_open else "proved", + } + + +def _contract(*articles: dict[str, object], open_statements: object = True) -> dict[str, object]: + return { + "articles": list(articles), + "open_statements": open_statements, + "schema": "autoform-assumptions/v1", + "source_revision": "fixture", + } + + +def _contract_file(path: Path, contract: object) -> Path: + path.write_text(json.dumps(contract), encoding="utf-8") + return path + + +def test_contract_lists_each_article_declaration(helper: ModuleType, tmp_path: Path) -> None: + contract = _contract_file( + tmp_path / "contract.json", + _contract( + _article("open", ["Fixture.open_stmt"], is_open=True, allowed=["Fixture.open_stmt"]), + _article("uses", ["Fixture.uses", "Fixture.uses"], allowed=["Fixture.open_stmt"]), + ), + ) + + entries = helper.load_assumption_contract(contract) + + assert [(entry.name, entry.article, entry.is_open, entry.allowed) for entry in entries] == [ + ("Fixture.open_stmt", "open", True, ("Fixture.open_stmt",)), + ("Fixture.uses", "uses", False, ("Fixture.open_stmt",)), + ] + + +@pytest.mark.parametrize( + ("contract", "message"), + [ + ("not json", "cannot read the assumption contract"), + ('{"schema": "a", "schema": "b"}', "duplicate key"), + ('{"value": NaN}', "unsupported JSON constant"), + ([], "not an autoform-assumptions/v1 object"), + ({**_contract(), "schema": "autoform-assumptions/v2"}, "not an autoform-assumptions/v1 object"), + (_contract(open_statements=False), "does not allow open statements"), + (_contract(open_statements="true"), "does not allow open statements"), + ({**_contract(), "articles": {}}, "no articles list"), + (_contract("article"), "not an object"), # type: ignore[arg-type] + (_contract(_article("", ["Fixture.a"])), "invalid article id"), + (_contract(_article("bad\nid", ["Fixture.a"])), "invalid article id"), + (_contract(_article("a", ["Fixture.a"]), _article("a", ["Fixture.b"])), "lists a twice"), + (_contract({**_article("a", ["Fixture.a"]), "open": "no"}), "malformed entry for a"), + (_contract(_article("a", [])), "malformed entry for a"), + (_contract({**_article("a", ["Fixture.a"]), "allowed_open_declarations": "x"}), "malformed entry"), + (_contract({**_article("a", ["Fixture.a"]), "assumes": [1]}), "malformed entry for a"), + (_contract({**_article("a", ["Fixture.a"]), "state": None}), "malformed entry for a"), + (_contract({**_article("a", ["Fixture.a"]), "article_id": 7}), "malformed entry for a"), + (_contract(_article("a", ["Fixture..a"])), "invalid Lean declaration"), + (_contract(_article("a", ["Fixture.«a"])), "invalid Lean declaration"), + (_contract(_article("a", ["Fixture.a b"])), "invalid Lean declaration"), + (_contract(_article("a", [7])), "invalid Lean declaration"), # type: ignore[list-item] + (_contract(_article("a", ["Fixture.a"], allowed=["Fixture.\u0000"])), "invalid Lean declaration"), + ( + _contract( + _article("open", ["Fixture.s"], is_open=True, allowed=["Fixture.s"]), + _article("proved", ["Fixture.s"]), + ), + "Fixture.s is an open statement of open but proved records it as proved", + ), + ( + _contract(_article("uses", ["Fixture.uses"], allowed=["Fixture.unknown"])), + "lets uses rest on Fixture.unknown, which no open article declares", + ), + ], +) +def test_contract_validation_fails_closed( + helper: ModuleType, tmp_path: Path, contract: object, message: str +) -> None: + path = tmp_path / "contract.json" + if isinstance(contract, str): + path.write_text(contract, encoding="utf-8") + else: + _contract_file(path, contract) + + with pytest.raises(helper.AuditInputError, match=message): + helper.load_assumption_contract(path) + + +def test_open_probe_spells_names_as_components(helper: ModuleType, tmp_path: Path) -> None: + contract = _contract_file( + tmp_path / "contract.json", + _contract( + _article("open", ["Fixture.«a.b c»"], is_open=True, allowed=["Fixture.«a.b c»"]), + _article('odd "id" \\', ["Fixture.x.1"], allowed=["Fixture.«a.b c»"]), + ), + ) + + probe = helper.render_open_probe(("Fixture",), helper.load_assumption_contract(contract)) + + table = next(line for line in probe.splitlines() if "readArticles " in line and " with" in line) + literal = table.split("readArticles ", 1)[1].rsplit(" with", 1)[0] + assert json.loads(json.loads(literal)) == [ + [["Fixture", "a.b c"], "open", True, [["Fixture", "a.b c"]]], + [["Fixture", "x", 1], 'odd "id" \\', False, [["Fixture", "a.b c"]]], + ] + assert probe.startswith("import Fixture\n") + assert "kernel trust clean except declared open statements" in probe + assert "kernel trust clean (" not in probe + with pytest.raises(helper.AuditInputError, match="empty open-statement audit"): + helper.render_open_probe((), ()) + + +def test_helper_cli_forms(repo_root: Path, tmp_path: Path) -> None: + helper_path = str(repo_root / _TEMPLATE) + blueprint = tmp_path / "blueprint" + _roadmap(blueprint, "---\nopen_statements: allowed\n---\n") + + def run(*arguments: str) -> subprocess.CompletedProcess[str]: + return subprocess.run( + [sys.executable, helper_path, *arguments], capture_output=True, text=True + ) + + policy = run("--policy", str(blueprint)) + assert policy.returncode == 0, policy.stderr + assert policy.stdout == "allowed\n" + _roadmap(blueprint, "---\nopen_statements: sometimes\n---\n") + refused = run("--policy", str(blueprint)) + assert refused.returncode == 1 + assert refused.stderr.startswith("error: ") and "accepts allowed or forbidden" in refused.stderr + + archive = _archive(tmp_path / "root.tgz", _module_members("Fixture")) + contract = _contract_file( + tmp_path / "contract.json", + _contract(_article("open", ["Fixture.s"], is_open=True, allowed=["Fixture.s"])), + ) + probe = tmp_path / "probe.lean" + prepared = run("--open-statements", str(contract), "Fixture", str(archive), str(probe)) + assert prepared.returncode == 0, prepared.stderr + assert prepared.stdout == ( + "prepared open-statement audit for 1 root-package module(s) and 1 open statement candidate(s)\n" + ) + assert "AutoformOpenStatementAudit.readArticles" in probe.read_text(encoding="utf-8") + + _contract_file(contract, _contract(open_statements=False)) + forbidden = run("--open-statements", str(contract), "Fixture", str(archive), str(probe)) + assert forbidden.returncode == 1 + assert "does not allow open statements" in forbidden.stderr + + for arguments in ( + (), + ("--policy",), + ("--policy", "a", "b"), + ("--open-statements", "contract", "Fixture", "archive"), + ("--open-statements", "contract", "Fixture", "archive", "probe", "extra"), + ): + usage = run(*arguments) + assert usage.returncode == 2, arguments + assert usage.stderr.count("autoform_audit.py") == 4 + assert "--open-statements CONTRACT ROOT_PACKAGE ROOT_BUILD_ARCHIVE OUTPUT_PROBE" in usage.stderr + + +_DEPENDENCY_LEAN = """theorem Dep.dep_sorry : True := sorry +theorem Dep.dep_clean : True := trivial +""" + +_OPEN_LEAN = """import Dep + +namespace Fixture + +theorem open_stmt : 1 + 1 = 2 := sorry + +theorem reduction : 1 + 1 = 2 ∧ True := ⟨open_stmt, trivial⟩ + +theorem clean : True := Dep.dep_clean + +end Fixture +""" + +_BAD_LEAN = _OPEN_LEAN + """ +namespace Fixture + +theorem helper_sorry : True := sorry + +theorem where_sorry : True := aux +where aux : True := sorry + +theorem type_sorry : (sorry : Prop) := sorry + +theorem uses_dependency_sorry : True := Dep.dep_sorry + +end Fixture +""" + + +@pytest.fixture(scope="module") +def open_projects(tmp_path_factory: pytest.TempPathFactory) -> dict[str, tuple[Path, Path]]: + if shutil.which("lake") is None: + pytest.skip("Lake is not installed") + root = tmp_path_factory.mktemp("open-statements") + dependency = root / "dependency" + _write(dependency / "lean-toolchain", "leanprover/lean4:v4.32.2\n") + _write(dependency / "lakefile.toml", 'name = "Dep"\ndefaultTargets = ["Dep"]\n\n[[lean_lib]]\nname = "Dep"\n') + _write(dependency / "Dep.lean", _DEPENDENCY_LEAN) + projects: dict[str, tuple[Path, Path]] = {} + for name, source in (("open", _OPEN_LEAN), ("bad", _BAD_LEAN)): + project = root / name + _write(project / "lean-toolchain", "leanprover/lean4:v4.32.2\n") + _write( + project / "lakefile.toml", + 'name = "Fixture"\ndefaultTargets = ["Fixture"]\n\n' + '[[require]]\nname = "Dep"\npath = "../dependency"\n\n' + '[[lean_lib]]\nname = "Fixture"\n', + ) + _write(project / "Fixture.lean", source) + built = _run(project, "lake", "build") + assert built.returncode == 0, built.stdout + built.stderr + archive = project / "root.tgz" + packed = _run(project, "lake", "pack", str(archive)) + assert packed.returncode == 0, packed.stdout + packed.stderr + projects[name] = (project, archive) + return projects + + +def _audit( + helper: ModuleType, project: tuple[Path, Path], contract: dict[str, object] | None +) -> subprocess.CompletedProcess[str]: + directory, archive = project + modules = helper.modules_from_archive(archive, "Fixture") + if contract is None: + text = helper.render_probe(modules) + else: + path = _contract_file(directory / "contract.json", contract) + text = helper.render_open_probe(modules, helper.load_assumption_contract(path)) + probe = directory / "probe.lean" + probe.write_text(text, encoding="utf-8") + return _run(directory, "lake", "env", "lean", str(probe)) + + +_OPEN_ARTICLE = _article("open", ["Fixture.open_stmt"], is_open=True, allowed=["Fixture.open_stmt"]) + + +def test_open_probe_accepts_declared_open_statements_and_reductions( + helper: ModuleType, open_projects: dict[str, tuple[Path, Path]] +) -> None: + audited = _audit( + helper, + open_projects["open"], + _contract( + _OPEN_ARTICLE, + _article("reduction", ["Fixture.reduction"], allowed=["Fixture.open_stmt"]), + _article('clean "odd" \\ id', ["Fixture.clean"]), + _article("dependency", ["Dep.dep_clean"]), + ), + ) + + output = audited.stdout + audited.stderr + assert audited.returncode == 0, output + assert "open statement (proof is sorry): Fixture.open_stmt [open]" in output + assert "conditional: Fixture.reduction [reduction] rests on open statement(s) Fixture.open_stmt" in output + assert 'sorry-free: Fixture.clean [clean "odd" \\ id]' in output + assert "sorry-free: Dep.dep_clean [dependency]" in output + assert "kernel trust clean except declared open statements (" in output + assert "; 1 open statement(s), 1 conditional declaration(s))" in output + assert "kernel trust clean (" not in output + + +@pytest.mark.parametrize( + ("articles", "message"), + [ + ( + [_OPEN_ARTICLE, _article("reduction", ["Fixture.reduction"])], + "Fixture.reduction [reduction] rests on open statement(s) Fixture.open_stmt, " + "which its article's Markdown dependencies do not reach", + ), + ( + [_article("open", ["Fixture.open_stmt"])], + "Fixture.open_stmt contains sorry but is not an open statement", + ), + ], +) +def test_open_probe_rejects_sorry_the_markdown_does_not_declare( + helper: ModuleType, + open_projects: dict[str, tuple[Path, Path]], + articles: list[dict[str, object]], + message: str, +) -> None: + audited = _audit(helper, open_projects["open"], _contract(*articles)) + + output = audited.stdout + audited.stderr + assert audited.returncode != 0, output + assert message in output + assert "root-package declarations failed the open-statement audit" in output + + +def test_open_probe_rejects_sorry_outside_an_open_statement_body( + helper: ModuleType, open_projects: dict[str, tuple[Path, Path]] +) -> None: + audited = _audit( + helper, + open_projects["bad"], + _contract( + _OPEN_ARTICLE, + _article("where", ["Fixture.where_sorry"], is_open=True, allowed=["Fixture.where_sorry"]), + _article("type", ["Fixture.type_sorry"], is_open=True, allowed=["Fixture.type_sorry"]), + _article("missing", ["Fixture.missing"]), + ), + ) + + output = audited.stdout + audited.stderr + assert audited.returncode != 0, output + for message in ( + "Fixture.helper_sorry contains sorry but is not an open statement", + "Fixture.where_sorry.aux contains sorry but is not an open statement", + "Fixture.type_sorry has sorry in its statement", + "Fixture.uses_dependency_sorry uses Dep.dep_sorry, which is outside the root package and depends on sorry", + "Fixture.missing [missing] is not a declaration of the Lean build", + "root-package declarations failed the open-statement audit", + ): + assert message in output + + +def test_strict_probe_still_rejects_every_sorry( + helper: ModuleType, open_projects: dict[str, tuple[Path, Path]] +) -> None: + audited = _audit(helper, open_projects["open"], None) + + output = audited.stdout + audited.stderr + assert audited.returncode != 0, output + assert "Fixture.open_stmt depends on unexpected axiom sorryAx" in output + assert "root-package declarations failed the kernel-trust audit" in output From a2ae5873a9427e6523ecd1e030507d02d74c8402 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 07:50:44 -0400 Subject: [PATCH 10/38] Document open statements and the revision contract Describe the open_statements policy, the conditional state, work assumptions, work impact, multi-key claims, the CI open-statement audit and its pin requirement, and the revision contract in the CLI reference, and point the Formalize, Roadmap, Human Review, and Agent Review skills at them. --- autoform_cli/README.md | 231 +++++++++++++++++- skills/agent-review/SKILL.md | 4 +- .../references/proof-integrity.md | 17 +- skills/formalize/SKILL.md | 44 +++- skills/human-review/SKILL.md | 12 +- skills/roadmap/SKILL.md | 22 +- 6 files changed, 299 insertions(+), 31 deletions(-) diff --git a/autoform_cli/README.md b/autoform_cli/README.md index 0aff4388..dabae359 100644 --- a/autoform_cli/README.md +++ b/autoform_cli/README.md @@ -90,20 +90,34 @@ An article asserts only facts a human or agent verified: | `lean: Ns.decl` | Declaration name(s) that discharge the article. | | `discussion: 42` | Issue number or URL where the article is being discussed. | | `article_id: af_...` | Durable identity, `af_` plus 24 lowercase hex digits; `autoform work` requires it on unfinished formalizable leaves. | +| `open_statements: allowed` | Project policy, valid only in `roadmap/README.md`: a theorem's statement may land with a `sorry` proof (see [Open statements](#open-statements)). Absent or `forbidden` keeps the strict policy. | Everything a reader thinks of as progress is *derived* from the DAG on every -run, so it cannot go stale: +run, so it cannot go stale. Readiness depends on the project's policy: under +the default strict policy CI rejects every `sorry`, so a theorem's statement +lands only with its proof; under `open_statements: allowed` it may land with a +`sorry` proof. | Derived state | Holds when | | --- | --- | -| `can_state` | Every statement prerequisite is stated. | -| `can_prove` | Stated, and every proof prerequisite is proved. | +| `can_state` | Every statement prerequisite is stated and, under the strict policy, every proof prerequisite is proved. | +| `can_prove` | Stated, every statement prerequisite is stated, and every proof prerequisite is proved (strict policy) or stated (open policy). | | `proved` | The proof compiles. | +| `conditional` | Proved, but the proof rests on an open statement; open policy only. | | `fully_proved` | Proved, and every prerequisite is fully proved, recursively. | | `defined` | A definition is written but rests on unfinished work. | `proved` and `fully_proved` differ on purpose: a theorem whose own proof -compiles but which rests on an unproved lemma is green, not dark green. The +compiles but which rests on unfinished work is green, not dark green. +`conditional`, labelled "conditionally proved", is the open policy's case of +that: the article records `proof: formalized` and its proof compiles, but it +reaches an open statement, an article whose statement is formalized and whose +proof is not, through its dependencies. An open dependency counts together with +whatever its statement prerequisites reach, and a proved dependency passes on +everything it reaches. The site colours it violet, never green, and lists those +open statements in an `Assumes` row on the article page. It is never +`fully_proved`, which keeps its meaning in both policies; a conditional result +is complete only once every open statement it assumes is proved. The palette and state names follow [leanblueprint](https://pypi.org/project/leanblueprint/), so the published graph reads the same way as the Lean community's LaTeX blueprints. @@ -496,10 +510,13 @@ autoform work context chapter/result . --lean-root . --json ``` `work list` returns only formalizable leaves whose next statement or proof phase -is unblocked. Project CI rejects `sorry`, so a theorem's statement lands with -its proof, and an article's statement phase also waits until its `## Proof -depends on` prerequisites are proved. The derived `can_state` state and the -site's Next up card do not apply this gate; dispatch from `work list`. `work +is unblocked, by the same policy-dependent rule as the derived `can_state` and +`can_prove` states, the runtime projection, and the site's Next up card. Under +the strict policy project CI rejects `sorry`, so a theorem's statement lands +with its proof, and an article's statement phase also waits until its `## Proof +depends on` prerequisites are proved. Under `open_statements: allowed` the +statement phase waits only for the statement prerequisites to be stated, and +the proof phase for every prerequisite to be stated. `work context` accepts the path-derived node ID (see Articles and containment) or an assigned `article_id` and reports the exact article, dependencies, source targets, Lean targets, blockers, article and graph source revisions, and claim @@ -517,6 +534,64 @@ frontmatter. `work context` may still select that article by its path ID to report the migration blocker. Both commands are read-only projections of Markdown. +Under the open policy the text output of `work list` starts with an `Open +statements: allowed` line and adds an `assumes:` line under each item that rests +on open statements; `work context` prints `Open statements:` and `Assumes:` +lines. In JSON, the frontier and every item carry `open_statements`, and every +item carries `assumes`, the open statements its proof rests on (empty under the +strict policy). The runtime projection carries the same `open_statements` flag, +and each node's runtime status adds `assumes` and `waiting_on`, the +prerequisites that keep an unproved node from its next phase. + +List the open statements and the articles that rest on them: + +```bash +autoform work assumptions . +autoform work assumptions blueprint --json +``` + +`work assumptions` prints the policy, one `open:` line per open statement with +its declarations, and one `conditional:` line per article whose Lean rests on +open statements. `--json` writes the `autoform-assumptions/v1` contract that CI +audits the build against: every stated article whose `lean:` names a +declaration, with `open`, `assumes`, and `allowed_open_declarations`, the +declarations of the open statements its Lean may reach, plus its own when it is +open. Under the strict policy every such article is listed with `open` false +and nothing allowed. It reads Markdown only and needs no Lean build. + +Ask what revising an article's Lean declarations would affect before editing +them: + +```bash +autoform work impact chapter/result . --lean-root . +autoform work impact chapter/result . --lean-root . --declaration MyProject.helper --json +``` + +The revised set is the article's `lean:` declarations, or the `--declaration` +names, each of which must be a project-local constant. `work impact` runs a +Lean probe against the built project through the same freshness check, output +bound, and process handling as `skeleton`, so it needs a fresh `lake build` +and a writable `.lake`, and it carries the same trust caveat: run it only in a +trusted checkout or a sandbox. `--timeout SECONDS` sets the probe's budget, +600 seconds by default. The report lists: + +- statement-impacted articles, whose declarations' meaning (a type, or a + definition's body or an inductive's constructors) reaches the revised set; +- proof-impacted articles, whose proofs or definition bodies use the revised + set, directly or through Lean-generated companions such as the `_simp_1` + lemma `simp` uses, without their meaning changing; +- helpers that no article names, each with an owner when one exists: the + article naming its nearest ancestor by name, such as `Foo` for `Foo.aux`; +- impacted articles without a Markdown dependency path to the revised article; +- every deprecated project declaration with its replacement and users, and in + `deprecated_unused` those nothing uses; +- the claim targets: the revised article's first, then every impacted + article's. + +A revision nothing else uses is `contained` and can be made in place. `--json` +writes `autoform-impact/v1`. The [revision contract](#revision-contract) says +what to do with the answer. + Plan durable article identity metadata without changing the blueprint: ```bash @@ -537,6 +612,7 @@ export AUTOFORM_WORKER_ID="agent-name" autoform claim acquire af_5b0e4d3c2a1f09e8d7c6b5a4 autoform claim renew af_5b0e4d3c2a1f09e8d7c6b5a4 autoform claim release af_5b0e4d3c2a1f09e8d7c6b5a4 +autoform claim acquire af_5b0e4d3c2a1f09e8d7c6b5a4 af_0123456789abcdef01234567 ``` Claim an article by the `claim_target` that `work context` reports. The board @@ -547,7 +623,8 @@ state does not persist between commands, as in agent tool calls, pass it with `--worker-id` on every command instead of exporting `AUTOFORM_WORKER_ID` once. Leases expire after 1500 seconds unless `--ttl` sets another length. Renew well within that, and confirm a claim is still held with `renew`, not `acquire`, -which also succeeds once a lease has expired or been released. +which also succeeds once a lease has expired or been released. Several targets +in one command change all-or-nothing; see the [claim contract](#claim-contract). Claims are fail-closed compare-and-swap leases under `refs/autoform-claims/` on the Git `origin`; pass `--repo` for another claim @@ -640,6 +717,86 @@ The audit API also accepts an already compiled graph. Formalize may use its findings while working the Markdown frontier, but the audit itself never enqueues work, stamps articles, or creates another graph artifact. +## Open statements + +By default a project runs the strict policy: CI rejects every `sorry`, so a +theorem's statement lands only together with its proof, and a statement waits +until the proof's prerequisites are proved. `open_statements: allowed` in +`roadmap/README.md` lets a theorem's statement land with a `sorry` proof. Such +an article, statement formalized and proof not, is an open statement. A proof +that uses open statements is conditional: it compiles and records `proof: +formalized`, but status, `work`, the site, and CI report it as conditionally +proved, never as fully proved. The policy lets dependents be stated and proved +against a faithful statement before its proof exists; the price is conditional +results that stay incomplete until every open statement they rest on is proved. + +Write an open statement's proof as exactly `sorry`. The audit accepts a `sorry` +only inside the proof of a theorem that an open article's `lean:` names: never +in its type, a helper, a definition, or a `where` clause, and never inherited +from a declaration outside the root package. Lean-generated auxiliaries count as +helpers: a `where` clause becomes `T.aux`, and well-founded recursion over two +or more arguments moves the `decreasing_by` proof into `T._unary`, so both +fail. `lake build --wfail` and `warningAsError` turn Lean's "declaration uses +`sorry`" warning into an error, so they cannot be combined with open +statements; the generated workflow runs plain `lake build`. + +The generated `autoform-verify.yml` reads the policy with `python3 +.github/autoform_audit.py --policy blueprint`, which prints `allowed` or +`forbidden` from `roadmap/README.md` (`forbidden` when the file or key is +absent) and exits 1 with `error: ...` on malformed frontmatter. Under +`forbidden` the audit is unchanged: any `sorryAx` fails with `NAME depends on +unexpected axiom sorryAx` and `root-package declarations failed the +kernel-trust audit`. Under `allowed` the workflow writes the `autoform work +assumptions blueprint --json` contract and audits every root-package +declaration against it. The audit accepts a `sorry` in a declared open +statement's own proof and a proof that reaches only the open statements its +article's Markdown dependencies reach. It rejects a `sorry` in a statement, a +`sorry` anywhere else in the root package, a dependency outside the root package +that depends on `sorry`, an article declaration that reaches an open statement +its Markdown dependencies do not reach (so a fully proved article, which +assumes nothing, may reach none), a `lean:` name missing from the build, +and an open statement that its article records as proved. Each article +declaration gets one log line, with `NAME` the declaration and `ID` the +article's node ID: + +```text +open statement (proof is sorry): NAME [ID] +open statement (proof depends on sorry elsewhere): NAME [ID] +open statement (proof is sorry-free; record proof: formalized): NAME [ID] +conditional: NAME [ID] rests on open statement(s) A, B +sorry-free: NAME [ID] +``` + +A passing audit ends with `kernel trust clean except declared open statements +(N root-package declaration(s) audited; K open statement(s), C conditional +declaration(s))`; a failing one logs each error, naming the declaration and +what to change, and ends with `root-package declarations failed the +open-statement audit`. + +The step runs `autoform work assumptions` from `AUTOFORM_REF`, and the earlier +`autoform check` step validates the frontmatter with that same pin. Scaffolded +workflows pin `AUTOFORM_REF` to the Autoform checkout that scaffolded them, so a +project scaffolded before open statements existed must, before opting in, move +`AUTOFORM_REF` to a commit that has `work assumptions` and replace +`.github/workflows/autoform-verify.yml` and `.github/autoform_audit.py` with the +versions `autoform init` writes at that commit. Both halves fail closed: an +older pin stops at `autoform check` with `unsupported frontmatter key +'open_statements'`, and an older workflow runs the strict audit, which rejects +every `sorry`. + +To reproduce the open-statement audit locally after a build, with +`ROOT_PACKAGE` the Lake package name the workflow reads from `lake +translate-config toml`: + +```bash +lake build +lake pack /tmp/autoform-root.tgz +autoform work assumptions blueprint --json > /tmp/autoform-assumptions.json +python3 .github/autoform_audit.py --open-statements /tmp/autoform-assumptions.json \ + ROOT_PACKAGE /tmp/autoform-root.tgz /tmp/autoform-probe.lean +lake env lean /tmp/autoform-probe.lean +``` + ## Claim contract Claims use canonical `autoform-claim/v1` JSON in orphan commit messages and @@ -655,10 +812,66 @@ worktree each and serialize `lake build` behind a `lake-build` claim, because builds share the elan toolchain and the Mathlib cache even when the checkouts are separate. +`acquire`, `renew`, and `release` accept several targets. The board reads +every ref once, applies the single-key ownership checks to each, and sends one +atomic push with a lease per ref, so either every claim changes or none does; +a board that cannot push atomically is refused. Success prints the usual line +per target. A refusal exits 1 with one line such as `error: could not acquire +af_y, af_z; no claim was acquired: held by another worker: af_y`, naming the +blocking targets when known, and leaves no claim behind; a duplicate target +exits 2. Workers that need several +claims follow a no-hold-and-wait rule: acquire the whole set in one command and, +when it is refused, release everything already held and retry with the whole +set, never holding some claims while waiting for others. The CLI does not +enforce the rule; it is what keeps two multi-article revisions from +deadlocking. + Claims are temporary operational state, never article frontmatter. Future Deicyde workers may share this protocol, but their current continue-uncoordinated failure behavior must be removed before they use the canonical claim API. +## Revision contract + +Revising a declaration X of article R that other articles' Lean uses touches +work R's claim does not cover. This contract makes that work claimable and +keeps the default build passing. + +1. On a fresh build, run `autoform work impact R . --lean-root .`, with + `--declaration` when only some of R's declarations, or a helper, change. +2. Choose the route: + - **Contained** (`contained: true`): revise X in place under R's claim. + - **Expand, migrate, contract**, the default whenever anything uses X: add + X' with the revised statement, leave X unchanged and mark it + `@[deprecated X' (since := "YYYY-MM-DD")]`, and point R's `lean:` at X'. + Statement-impacted articles lose `statement` and `proof` but keep `lean:`, + so they return to the frontier. Proof-impacted articles keep everything, + since their proofs still use the valid old X; migrating them to X' is + later work. Claim R and the statement-impacted articles, whose + frontmatter changes. + - **In place**, only when X and X' cannot coexist, for example an instance + or a structure change: claim every `claim_targets` entry plus the claim + target of every helper's owner, and repair every impacted declaration in + one commit whose default build passes. A repaired dependent proof keeps + `proof: formalized` only after an Agent Review of the repair; otherwise + retract it. Under the open policy, replace its proof with `sorry` and + remove `proof`; under the strict policy, remove the declaration and its + assertions only if nothing else uses it, and use expand, migrate, + contract otherwise. Record what happened under `## Execution notes` of + each touched article. +3. Claim the whole set with one `autoform claim acquire`. When it is refused, + release everything and report the held claim as the blocker. After + acquiring, re-run `work impact`; if the set grew, release and start over + with the larger set. Land one commit, then release every claim. +4. Contract: delete a deprecated X once `work impact R . --lean-root . + --declaration X` reports it contained, or X appears in `deprecated_unused`. +5. `autoform audit --lean-root` reports `lean-target-deprecated` for an article + whose `lean:` names a declaration carrying the `deprecated` attribute; point + it at the replacement. +6. When Roadmap revises an article's statement text, it removes that article's + `statement` and `proof` but keeps `lean:`, which `work impact` needs. It + retracts only that article and the dependents whose Markdown text the + revision rewrites; the Lean-side impact decides every other dependent. + ## Local runtime doctor Use the runtime projection and roadmap audit together without contacting any diff --git a/skills/agent-review/SKILL.md b/skills/agent-review/SKILL.md index a5b9e4cc..9f8fe537 100644 --- a/skills/agent-review/SKILL.md +++ b/skills/agent-review/SKILL.md @@ -24,7 +24,9 @@ Select the rubric from the artifact under review. Keep objective evidence separate from judgment. Never claim compilation, declaration resolution, axiom cleanliness, source coverage, or dependency correctness without showing how it was checked. If required sources are absent, -return insufficient evidence rather than guessing. +return insufficient evidence rather than guessing. In a project that allows +open statements, a proof resting on declared open statements is conditional: +name the statements it assumes and never call it axiom-clean. Regenerate skeleton evidence from the exact candidate after its Lean build. Do that only in a trusted checkout or an operating-system sandbox: the command diff --git a/skills/agent-review/references/proof-integrity.md b/skills/agent-review/references/proof-integrity.md index 1bbd36a5..825a0eaf 100644 --- a/skills/agent-review/references/proof-integrity.md +++ b/skills/agent-review/references/proof-integrity.md @@ -14,6 +14,21 @@ is insufficient when its dependencies are hollow. infrastructure does not justify axiomatizing content that the source proves. 5. In a repository with an explicit audited axiom ledger, apply that repository's ledger and discharge policy in addition to this rubric. +6. If `roadmap/README.md` sets `open_statements: allowed`, list the open statements the reviewed + article may assume (`autoform work assumptions`) and trace each `sorryAx` to its source, for + example with the CI audit the [open statements reference](../../../autoform_cli/README.md#open-statements) + reproduces, which logs each declaration as open, conditional, or sorry-free. + +## Open statements + +Under the open-statement policy, a `sorry` inside a declared open upstream statement, one the +reviewed article reaches through its Markdown dependencies and that `work assumptions` attributes +to it, is a recorded dependency of the reviewed declaration, not a gap in it. Score the reviewed +proof on its own work, name every assumed open statement in the review, and mark the verdict as due +for re-review when one is discharged, since the skeleton's axioms and hashes then move. A `sorry` +anywhere else, including a helper, a `where` clause, or the reviewed declaration's own statement, or +an open statement that is not a declared dependency, keeps the ceilings below. A conditional proof +is never axiom-clean, complete, or sorry-free, whatever its score. ## Structural red flags @@ -42,4 +57,4 @@ the instantiated theorem really implies the reviewed statement. Hard ceilings: any axiom or sorry covering source-proved content scores at most 1; orphan classes, vacuous definitions, or trivial instances score at most 2. Give a separate justified/unjustified verdict for every nonstandard axiom, sorry, or structural issue found. Pass at 3; reject at 2 or -below, and never describe a result with a score-3 gap as axiom-clean. +below, and never describe a result with a score-3 gap, or a conditional proof, as axiom-clean. diff --git a/skills/formalize/SKILL.md b/skills/formalize/SKILL.md index 7226a85c..0998f4d3 100644 --- a/skills/formalize/SKILL.md +++ b/skills/formalize/SKILL.md @@ -60,18 +60,40 @@ articles may legitimately change the graph-wide source revision. Read the complete article, cited sources, dependency articles, and existing Lean target. Preserve the exact mathematical statement. Work only on the selected phase: do not modify another article or its Lean declarations, or weaken a -public statement. Search the pinned Mathlib checkout before adding helpers, and -use the shared Lean LSP and REPL with `` as the project path. Finish -with the focused Lake target. Declare the result in a module the library root -imports, or that the lakefile's globs cover, because the default build is the -only one CI compiles and audits. Check the recorded declaration with `#print -axioms`; do not accept `sorry`, new axioms, unsafe shortcuts, a weaker theorem, -unused hypotheses, or an unrelated declaration. +public statement. The one exception is revising a declaration other articles' +Lean uses: start from `autoform work impact` and make only the edits the +[revision contract](../../autoform_cli/README.md#revision-contract) requires, +under the claims it requires. Search the pinned Mathlib checkout before adding +helpers, and use the shared Lean LSP and REPL with `` as the project +path. Finish with the focused Lake target. Declare the result in a module the +library root imports, or that the lakefile's globs cover, because the default +build is the only one CI compiles and audits. Check the recorded declaration +with `#print axioms`; do not accept `sorry`, new axioms, unsafe shortcuts, a +weaker theorem, unused hypotheses, or an unrelated declaration, except as the +open-statement policy below allows. -The statement phase writes the declaration. Project CI rejects `sorry`, so for a -theorem it also writes the complete proof; that is why `work list` offers the -phase only once the proof prerequisites are proved. The proof phase completes -the proof of an already recorded statement without changing that statement. +`roadmap/README.md` sets the project's policy. Under the default strict policy, +project CI rejects `sorry`: the statement phase writes the declaration and, for +a theorem, the complete proof, which is why `work list` offers the phase only +once the proof prerequisites are proved. Under `open_statements: allowed`, the +statement phase writes the faithful statement with a proof body of exactly +`sorry`, or the full proof, and records only `statement: formalized`. That +`sorry` goes in the declaration's own body, never in its type, a helper, a +definition, or a `where` clause. In both policies the proof phase completes the +proof of an already recorded statement without changing that statement. + +Under the open policy a proof may use the open statements of its declared +dependencies, direct or transitive. The article then shows as conditionally +proved, and `#print axioms` lists the `sorryAx` it inherits from them without +saying from where. `autoform work assumptions` lists the open statements the +Markdown lets this article assume. Before landing, record `proof: formalized` +in the worktree (until then the audit treats the article as open), reproduce +the CI audit as the [open statements +reference](../../autoform_cli/README.md#open-statements) shows, and require a +`conditional` or `sorry-free` line for each recorded declaration and a passing +summary. An open statement the Markdown does not declare as a dependency fails +CI: send the missing dependency to Roadmap or stop using it. Never describe a +conditional proof as complete, fully proved, or sorry-free. After the focused build passes, require an independent Agent Review of every changed statement or proof for source faithfulness, dependency correctness, and diff --git a/skills/human-review/SKILL.md b/skills/human-review/SKILL.md index 02e80a07..553c6e15 100644 --- a/skills/human-review/SKILL.md +++ b/skills/human-review/SKILL.md @@ -37,8 +37,9 @@ individual node and Lean-source links. Record each human decision as `approve`, `block`, with the exact page or node and rationale. Separate validator output from the person's judgment. Do not silently apply requested revisions: hand mathematical-plan changes and Lean implementation changes to Roadmap, which -reopens the affected articles for Formalize, and autonomous rubric scoring to -Agent Review. +records the decision and follows the [revision +contract](../../autoform_cli/README.md#revision-contract) for declarations +other articles use, and autonomous rubric scoring to Agent Review. Treat the landing page's `Scoped roadmap` percentage as completion among formalizable leaf targets that are fully proved, including every dependency @@ -48,4 +49,9 @@ is an author assertion, not audit verification that the declaration is in Mathlib. Treat the percentage never as whole-source completion. Read the adjacent declared source coverage and its linked coverage contract before making scope claims. A statement-only theorem remains incomplete whether it is -blocked or ready to prove. +blocked or ready to prove. In a project that allows open statements, a +conditionally proved target, whose proof rests on an open statement whose Lean +proof is still `sorry`, is not complete either: it is never fully proved, so it +counts toward the percentage's total but not its completed share, as does +every target that depends on it. Present it as conditional, naming the open +statements in its `Assumes` row, never as proved or sorry-free. diff --git a/skills/roadmap/SKILL.md b/skills/roadmap/SKILL.md index 3e912b5d..10e3735e 100644 --- a/skills/roadmap/SKILL.md +++ b/skills/roadmap/SKILL.md @@ -75,12 +75,22 @@ and release it once the committed revision is on the branch Formalize works from. A refused acquire means another agent owns the article: leave it and report it. Claims write refs to the board's remote, which is outward-facing, so make sure the request covers them. When a revision changes a statement, remove -the `statement`, `proof`, and `lean` metadata the new text no longer matches. -For a Lean revision requested in Human Review, record the decision in the -article and remove the assertions it invalidates there and on every dependent -whose Lean uses the changed declaration, so Formalize takes the work up from its -frontier. For a large source, divide independent sections among available agents -while retaining one owner for global coverage and dependency consistency. +its `statement` and `proof` metadata but keep `lean:`, which Formalize needs to +run `autoform work impact`. Retract only that article and the dependents whose +Markdown text the revision rewrites, claiming them all in one acquire; the +Lean-side impact decides every other dependent. For a Lean revision requested +in Human Review, record the decision in the article, then follow the [revision +contract](../../autoform_cli/README.md#revision-contract): run `work impact`, +claim the whole set it names atomically, and take the route it allows. For a +large source, divide independent sections among available agents while +retaining one owner for global coverage and dependency consistency. + +`open_statements` in `roadmap/README.md` is a project policy; absent means +strict. `allowed` lets a theorem's statement land with a `sorry` proof, so +dependents can be stated and proved before it is; the cost is conditionally +proved results that stay incomplete until those proofs land. Change it only on +the user's request and only once CI meets the +[open statements](../../autoform_cli/README.md#open-statements) requirements. Reconcile every affected source and milestone page, the coverage contract, `blueprint/README.md`, and the repository `README.md`. From 00af0857341cfe70a89747d94b42d7fe7dd79793 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:13:57 -0400 Subject: [PATCH 11/38] Keep a retracted theorem open while its lean: names the old declaration Retracting a statement removes `statement` but keeps `lean:`, so the old declaration and its sorry stay in the build until Formalize restates the article. Status and the assumption contract skipped unstated articles, which made CI reject that sorry as undeclared and showed every proof resting on it as plainly proved. Under the open policy such a theorem now stays an open statement, and every article with declarations, stated or not, gets a contract entry. --- autoform_cli/status.py | 14 +++++++++----- autoform_cli/work.py | 16 +++++++++++----- tests/test_status.py | 31 +++++++++++++++++++++++++++++++ tests/test_work.py | 24 ++++++++++++++++++++++++ 4 files changed, 75 insertions(+), 10 deletions(-) diff --git a/autoform_cli/status.py b/autoform_cli/status.py index 032076d5..7a6d8552 100644 --- a/autoform_cli/status.py +++ b/autoform_cli/status.py @@ -103,9 +103,10 @@ class NodeStatus: """The derived progress of a single node. ``waiting_on`` names the prerequisites that keep an unproved node from its - next phase, in authored order. ``assumes`` names the open statements (stated, - not proved) a proof of this node rests on; it is empty unless the project - allows open statements. + next phase, in authored order. ``assumes`` names the open statements (not + proved, but stated or a theorem still naming its ``lean:`` declaration) a + proof of this node rests on; it is empty unless the project allows open + statements. """ node_id: str @@ -176,13 +177,16 @@ def done(dependency_id: str, attribute: str) -> bool: can_state = not unstated can_prove = stated and not unstated and not unmet waiting = unstated + unmet if stated else unstated - # Mathlib and unstated nodes reach nothing, so they get no entry. + # Mathlib and unstated nodes reach nothing, so they get no entry, + # except a theorem whose statement a revision retracted while its + # `lean:` still names the old declaration: that declaration and its + # sorry stay in the build until Formalize restates it. if not node.mathlib: reached = frozenset().union(*(reaches.get(other, frozenset()) for other in node.dependencies)) assumes = tuple(sorted(reached)) if proved: reaches[node_id] = reached - elif stated: + elif stated or (node.lean and not definition): reaches[node_id] = frozenset({node_id}).union( *(reaches.get(other, frozenset()) for other in node.statement_dependencies) ) diff --git a/autoform_cli/work.py b/autoform_cli/work.py index 16a281d6..ada4b925 100644 --- a/autoform_cli/work.py +++ b/autoform_cli/work.py @@ -7,6 +7,7 @@ from pathlib import Path from .runtime import RuntimeNode, load_runtime_graph +from .status import is_definition WORK_SCHEMA = "autoform-work/v1" @@ -247,12 +248,13 @@ def to_json(self) -> str: def assumption_contract(project_or_blueprint: str | Path) -> AssumptionContract: - """Record which open statements each stated article's Lean code may reach. + """Record which open statements each article's Lean code may reach. CI audits the Lean build against this: an open article's own declarations may keep a ``sorry`` proof, and every other article may reach only the open - statements its Markdown dependencies declare. Under the strict policy no - article is open and nothing is allowed. + statements its Markdown dependencies declare. A theorem whose statement a + revision retracted stays open while its ``lean:`` names the old declaration. + Under the strict policy no article is open and nothing is allowed. """ runtime = load_runtime_graph(project_or_blueprint) declarations = { @@ -261,9 +263,13 @@ def assumption_contract(project_or_blueprint: str | Path) -> AssumptionContract: } articles: list[AssumptionArticle] = [] for node in sorted(runtime.nodes, key=lambda candidate: candidate.id): - if node.mathlib or not node.status.stated or not declarations[node.id]: + if node.mathlib or not declarations[node.id]: continue - is_open = runtime.open_statements and node.status.stated and not node.status.proved + is_open = ( + runtime.open_statements + and not node.status.proved + and (node.status.stated or not is_definition(node)) + ) allowed = {name for assumed in node.status.assumes for name in declarations.get(assumed, ())} if is_open: allowed.update(declarations[node.id]) diff --git a/tests/test_status.py b/tests/test_status.py index 68ee4195..d642863e 100644 --- a/tests/test_status.py +++ b/tests/test_status.py @@ -311,6 +311,37 @@ def test_assumptions_reach_what_the_lean_proof_can_reach(tmp_path: Path) -> None assert statuses["up"].assumes == () +@pytest.mark.parametrize("policy", ["forbidden", "allowed"]) +def test_a_retracted_theorem_still_naming_its_lean_stays_an_open_statement(tmp_path: Path, policy: str) -> None: + """A retracted statement's declaration, and any sorry in it, stay in the build until it is restated. + + A definition carries no sorry the audit would accept, so a retracted one is + not open. + """ + blueprint = tmp_path / "blueprint" + _node(blueprint, "README.md", open_statements=policy) + _node(blueprint, "retracted.md", declaration="theorem", lean="Ns.retracted") + _node(blueprint, "old.md", declaration="def", lean="Ns.old") + _node(blueprint, "gone.md", declaration="theorem") + _node( + blueprint, + "reduction.md", + "## Proof depends on\n\n- [Retracted](retracted.md)\n- [Old](old.md)\n- [Gone](gone.md)\n", + declaration="theorem", + statement="formalized", + proof="formalized", + lean="Ns.reduction", + ) + + statuses = derive(load_graph(blueprint)) + + assert statuses["retracted"].key == "can_state" + if policy == "allowed": + assert (statuses["reduction"].key, statuses["reduction"].assumes) == ("conditional", ("retracted",)) + else: + assert (statuses["reduction"].key, statuses["reduction"].assumes) == ("proved", ()) + + @pytest.mark.parametrize("open_statements", [False, True]) def test_a_missing_dependency_counts_as_neither_stated_nor_proved( tmp_path: Path, open_statements: bool diff --git a/tests/test_work.py b/tests/test_work.py index 916037ba..0e1e8cff 100644 --- a/tests/test_work.py +++ b/tests/test_work.py @@ -823,6 +823,30 @@ def test_work_assumptions_under_the_open_policy_bounds_each_article( ) +@pytest.mark.parametrize("policy", ["forbidden", "allowed"]) +def test_work_assumptions_keeps_a_retracted_theorem_while_its_lean_names_the_old_declaration( + tmp_path: Path, capsys, policy: str +) -> None: + """Roadmap retracts a statement but keeps `lean:`; the old sorry stays declared until Formalize restates it.""" + project = _policy_project(tmp_path, policy) + _edit(project, "open.md", "statement: formalized\n", "") + allowed = ("Project.open_aux", "Project.open_thm") if policy == "allowed" else () + + assert cli.main(["work", "assumptions", str(project), "--json"]) == 0 + articles = {article["id"]: article for article in json.loads(capsys.readouterr().out)["articles"]} + + assert articles["chapter/open"] == _contract_article( + "chapter/open", + "af_00000000000000000000000b", + "can_state", + ["Project.open_thm", "Project.open_aux"], + open_=policy == "allowed", + allowed=allowed, + ) + assert articles["chapter/reduction"]["state"] == ("conditional" if policy == "allowed" else "proved") + assert articles["chapter/reduction"]["allowed_open_declarations"] == list(allowed) + + def test_work_assumptions_reports_errors_on_stderr_with_exit_2( tmp_path: Path, capsys, monkeypatch: pytest.MonkeyPatch ) -> None: From fef8a4ea1a966f30ca549bb493c9b8127eb5d53a Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:13:57 -0400 Subject: [PATCH 12/38] Claim helper owners in work impact A helper is repaired under its owner's claim, so the owner's claim target belongs in claim_targets. Owners were reported only by node id, which is not the key the owner's worker claims when the article has an article_id. --- autoform_cli/impact.py | 5 ++++- tests/test_impact.py | 2 ++ 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/autoform_cli/impact.py b/autoform_cli/impact.py index 6c2dd1e1..132703e2 100644 --- a/autoform_cli/impact.py +++ b/autoform_cli/impact.py @@ -489,7 +489,10 @@ def compute_impact( if record.deprecated ) - claim_targets = (revised.claim_target, *sorted({item.claim_target for item in impacted} - {revised.claim_target})) + # A helper is repaired under its owner's claim, so the owner is claimed too. + owners = {by_id[helper.owner].claim_target for helper in helpers if helper.owner in by_id} + others = {item.claim_target for item in impacted} | owners + claim_targets = (revised.claim_target, *sorted(others - {revised.claim_target})) return ImpactReport( source_revision=source_revision, article=revised, diff --git a/tests/test_impact.py b/tests/test_impact.py index bcd7b3d6..374f4816 100644 --- a/tests/test_impact.py +++ b/tests/test_impact.py @@ -222,6 +222,8 @@ def locate(record: ConstantRecord) -> tuple[str | None, int | None]: ] assert located == [helper.name for helper in report.helpers] assert _ids(report.statement_impacted) == ["uses"] + # The helpers are repaired under their owners' claims, so shared-a is claimed too. + assert report.claim_targets == ("base", "shared-a", "uses") def test_revised_names_resolve_by_component_and_are_never_their_own_helpers() -> None: From 23c7f2c2c1752b3bd7a86d8c9aab496ed66a3f47 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:36:35 -0400 Subject: [PATCH 13/38] Describe assumed open statements without claiming they are sorry The Assumes row and the graph legend said a conditional proof rests on open statements whose Lean proofs are still sorry. A retracted theorem that keeps its lean: name counts as open even when its Lean proof is complete, and a sorry-free proof recorded with only its statement is open too, so both texts now say the open statement has no recorded Lean proof, which holds in every case. --- autoform_cli/mermaid.py | 2 +- autoform_cli/render.py | 2 +- tests/test_render.py | 2 +- tests/test_visualization.py | 2 +- 4 files changed, 4 insertions(+), 4 deletions(-) diff --git a/autoform_cli/mermaid.py b/autoform_cli/mermaid.py index 93d051a4..f34227e2 100644 --- a/autoform_cli/mermaid.py +++ b/autoform_cli/mermaid.py @@ -265,7 +265,7 @@ def render_legend_tip(statuses: dict[str, NodeStatus]) -> str: "fully_proved": "Proved, and every prerequisite is fully proved too.", "proved": "Proof compiles, but something it rests on is not finished.", "defined": "Definition is written in Lean.", - "conditional": "Proof compiles, but rests on an open statement whose Lean proof is still sorry.", + "conditional": "Proof compiles, but rests on an open statement without a recorded Lean proof.", "can_prove": "Statement is in Lean and nothing it needs is blocked, so the proof can start.", "stated": "Statement is in Lean; the proof is not.", "can_state": "Nothing it needs is blocked, so the statement can be written in Lean.", diff --git a/autoform_cli/render.py b/autoform_cli/render.py index 7891b9a4..01d0a3cd 100644 --- a/autoform_cli/render.py +++ b/autoform_cli/render.py @@ -1708,7 +1708,7 @@ def _render_environment( assumed = _node_references( node_status.assumes, graph=graph, statuses=statuses, numbers=numbers, links=links ) - meta_rows.append(("Assumes", f"{assumed} (open statements whose Lean proofs are still sorry)")) + meta_rows.append(("Assumes", f"{assumed} (open statements without a recorded Lean proof)")) if node.discussion: meta_rows.append(("Discussion", _discussion_link(node.discussion, linker))) meta = _render_rows(meta_rows, css_class="bp-meta") diff --git a/tests/test_render.py b/tests/test_render.py index cee93485..7fe72c57 100644 --- a/tests/test_render.py +++ b/tests/test_render.py @@ -1311,7 +1311,7 @@ def test_a_conditional_proof_names_the_open_statements_it_assumes(tmp_path: Path assert ( 'Assumes' 'Theorem 1 (Open)' - " (open statements whose Lean proofs are still sorry)" + " (open statements without a recorded Lean proof)" ) in top # Only the conditional proof carries the row; the open statement assumes nothing. assert page.count('Assumes') == 1 diff --git a/tests/test_visualization.py b/tests/test_visualization.py index 2d991aca..12bdf497 100644 --- a/tests/test_visualization.py +++ b/tests/test_visualization.py @@ -151,7 +151,7 @@ def test_a_conditional_proof_has_its_own_colour_and_legend_entry(tmp_path: Path) assert '("Top"):::conditional' in document assert f"classDef conditional fill:{_state('conditional').fill}" in document assert '' in document - assert "Proof compiles, but rests on an open statement whose Lean proof is still sorry." in document + assert "Proof compiles, but rests on an open statement without a recorded Lean proof." in document def test_the_legend_explains_readiness_without_naming_a_policy(tmp_path: Path) -> None: From 2bd633510da866f9fdadb0302c9937526fc29934 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:26:44 -0400 Subject: [PATCH 14/38] Hold a definition until its proof prerequisites are stated Under the open policy derive let a definition's statement wait only on its statement prerequisites, so work list, work context and the site offered a definition whose body uses an unstated theorem with no Lean declaration. A definition cannot land open: writing it down is its proof, and a sorry in a def fails CI. It now also waits for every proof prerequisite to be stated (an open statement counts), and waiting_on names the missing ones. The strict policy is unchanged. --- autoform_cli/status.py | 13 ++++++++----- tests/test_status.py | 29 +++++++++++++++++++++++++++++ tests/test_work.py | 21 +++++++++++++++++++++ 3 files changed, 58 insertions(+), 5 deletions(-) diff --git a/autoform_cli/status.py b/autoform_cli/status.py index 7a6d8552..4d197719 100644 --- a/autoform_cli/status.py +++ b/autoform_cli/status.py @@ -143,9 +143,10 @@ def derive(graph: Graph) -> dict[str, NodeStatus]: rejects every ``sorry``, so a theorem's statement lands only with its proof: both phases wait for the statement prerequisites to be stated and the proof prerequisites to be proved. When ``roadmap/README.md`` sets - ``open_statements: allowed``, a statement may land with a ``sorry`` proof, so - a statement waits only for its statement prerequisites and a proof for every - prerequisite to be stated. + ``open_statements: allowed``, a theorem's statement may land with a ``sorry`` + proof, so a statement waits only for its statement prerequisites and a proof + for every prerequisite to be stated. A definition has no ``sorry`` to land + with, so its statement waits for every prerequisite to be stated. """ open_policy = graph.open_statements statuses: dict[str, NodeStatus] = {} @@ -174,9 +175,11 @@ def done(dependency_id: str, attribute: str) -> bool: unmet.append(other) assumes: tuple[str, ...] = () if open_policy: - can_state = not unstated + # A definition cannot land open: writing it down is its proof, so + # its body uses its proof prerequisites, and they must be stated. + can_state = not unstated and not (definition and unmet) can_prove = stated and not unstated and not unmet - waiting = unstated + unmet if stated else unstated + waiting = unstated + unmet if stated or definition else unstated # Mathlib and unstated nodes reach nothing, so they get no entry, # except a theorem whose statement a revision retracted while its # `lean:` still names the old declaration: that declaration and its diff --git a/tests/test_status.py b/tests/test_status.py index d642863e..c269fd63 100644 --- a/tests/test_status.py +++ b/tests/test_status.py @@ -311,6 +311,35 @@ def test_assumptions_reach_what_the_lean_proof_can_reach(tmp_path: Path) -> None assert statuses["up"].assumes == () +@pytest.mark.parametrize("policy", ["forbidden", "allowed"]) +def test_a_definition_is_stated_only_once_its_proof_prerequisites_are(tmp_path: Path, policy: str) -> None: + """A definition's body is its proof, so it cannot land open with a missing prerequisite. + + ``gap`` is an unstated theorem and ``open`` a stated one with no proof. + Under the open policy ``open`` counts as stated, so only ``gap`` blocks. + """ + blueprint = tmp_path / "blueprint" + _node(blueprint, "README.md", open_statements=policy) + _node(blueprint, "gap.md", declaration="theorem") + _node(blueprint, "open.md", declaration="theorem", statement="formalized") + _node( + blueprint, + "blocked.md", + "## Proof depends on\n\n- [Gap](gap.md)\n- [Open](open.md)\n", + declaration="definition", + ) + _node(blueprint, "ready.md", "## Proof depends on\n\n- [Open](open.md)\n", declaration="definition") + + readiness = _readiness(derive(load_graph(blueprint))) + + if policy == "allowed": + assert readiness["blocked"] == ("planned", False, False, ("gap",), ("open",)) + assert readiness["ready"] == ("can_state", True, False, (), ("open",)) + else: + assert readiness["blocked"] == ("planned", False, False, ("gap", "open"), ()) + assert readiness["ready"] == ("planned", False, False, ("open",), ()) + + @pytest.mark.parametrize("policy", ["forbidden", "allowed"]) def test_a_retracted_theorem_still_naming_its_lean_stays_an_open_statement(tmp_path: Path, policy: str) -> None: """A retracted statement's declaration, and any sorry in it, stay in the build until it is restated. diff --git a/tests/test_work.py b/tests/test_work.py index 0e1e8cff..18690d97 100644 --- a/tests/test_work.py +++ b/tests/test_work.py @@ -616,6 +616,27 @@ def test_open_blockers_wait_only_for_prerequisites_to_be_stated(tmp_path: Path) assert "chapter/main" in {item.node_id for item in list_ready_work(project).items} +def test_open_work_holds_a_definition_until_its_proof_prerequisites_are_stated(tmp_path: Path) -> None: + """A definition cannot land open, so its body needs every prerequisite's declaration.""" + project = _policy_project(tmp_path, "allowed") + _article( + project, + "data.md", + title="Data", + metadata=["article_id: af_000000000000000000000013", "declaration: definition"], + proof_depends="state.md", + ) + + _, data = work_context(project, "chapter/data") + assert (data.phase, data.blockers) == (None, ("chapter/state",)) + assert "chapter/data" not in {item.node_id for item in list_ready_work(project).items} + + # An open statement is stated, so it is enough. + _edit(project, "data.md", "- [dependency](state.md)", "- [dependency](open.md)") + _, data = work_context(project, "chapter/data") + assert (data.phase, data.blockers, data.assumes) == ("statement", (), ("chapter/open",)) + + def test_strict_work_text_is_unchanged(tmp_path: Path, capsys) -> None: project = _policy_project(tmp_path, None) revision = load_runtime_graph(project).source_revision From 03bf0a489acbae838779b0782a29041fa842d984 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:27:39 -0400 Subject: [PATCH 15/38] Print open statements once in work assumptions The text output printed a conditional: line for every contract entry with assumptions, so an unproved open statement appeared as both open: and conditional:, while the docs reserve conditional for a proof that rests on an open statement. An open entry now carries its assumptions on its open: line, and conditional: is printed only for entries that are not open. The JSON contract is unchanged. --- autoform_cli/__main__.py | 9 ++++++--- tests/test_work.py | 3 +-- 2 files changed, 7 insertions(+), 5 deletions(-) diff --git a/autoform_cli/__main__.py b/autoform_cli/__main__.py index 2bbcf8ef..46f7b8dd 100644 --- a/autoform_cli/__main__.py +++ b/autoform_cli/__main__.py @@ -562,10 +562,13 @@ def _work_assumptions(args: argparse.Namespace) -> int: return 0 print(_human_text(f"Open statements: {'allowed' if contract.open_statements else 'forbidden'}")) for article in contract.articles: + assumes = f"assumes {', '.join(article.assumes)}" if article.open: - print(_human_text(f"open: {article.id} ({', '.join(article.declarations)})")) - if article.assumes: - print(_human_text(f"conditional: {article.id} assumes {', '.join(article.assumes)}")) + # An open statement is not conditional: its own proof is missing. + line = f"open: {article.id} ({', '.join(article.declarations)})" + print(_human_text(f"{line} {assumes}" if article.assumes else line)) + elif article.assumes: + print(_human_text(f"conditional: {article.id} {assumes}")) return 0 diff --git a/tests/test_work.py b/tests/test_work.py index 18690d97..a6a2684c 100644 --- a/tests/test_work.py +++ b/tests/test_work.py @@ -839,8 +839,7 @@ def test_work_assumptions_under_the_open_policy_bounds_each_article( "open: chapter/open (Project.open_thm, Project.open_aux)\n" "open: chapter/prove (Project.prove)\n" "conditional: chapter/reduction assumes chapter/open\n" - "open: chapter/uses (Project.uses)\n" - "conditional: chapter/uses assumes chapter/open\n" + "open: chapter/uses (Project.uses) assumes chapter/open\n" ) From 66d7cb9b9638c8027e19517d216cce9a1d210c91 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:28:14 -0400 Subject: [PATCH 16/38] Test the contract entry for a proof recorded without its statement An article with proof: formalized and no statement: formalized used to be skipped by assumption_contract, so its declarations escaped the CI open probe. Pin that it gets an entry that is not open and whose allowance is limited to the open statements it assumes, under both policies. --- tests/test_work.py | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/tests/test_work.py b/tests/test_work.py index a6a2684c..918ea282 100644 --- a/tests/test_work.py +++ b/tests/test_work.py @@ -867,6 +867,28 @@ def test_work_assumptions_keeps_a_retracted_theorem_while_its_lean_names_the_old assert articles["chapter/reduction"]["allowed_open_declarations"] == list(allowed) +@pytest.mark.parametrize("policy", ["forbidden", "allowed"]) +def test_work_assumptions_bounds_a_proof_recorded_without_its_statement( + tmp_path: Path, capsys, policy: str +) -> None: + """`proof: formalized` without `statement: formalized` still declares Lean the CI probe must bound.""" + project = _policy_project(tmp_path, policy) + _edit(project, "reduction.md", "statement: formalized\n", "") + allowed = ("Project.open_aux", "Project.open_thm") if policy == "allowed" else () + + assert cli.main(["work", "assumptions", str(project), "--json"]) == 0 + articles = {article["id"]: article for article in json.loads(capsys.readouterr().out)["articles"]} + + assert articles["chapter/reduction"] == _contract_article( + "chapter/reduction", + "af_00000000000000000000000c", + "conditional" if policy == "allowed" else "proved", + ["Project.reduction"], + assumes=("chapter/open",) if policy == "allowed" else (), + allowed=allowed, + ) + + def test_work_assumptions_reports_errors_on_stderr_with_exit_2( tmp_path: Path, capsys, monkeypatch: pytest.MonkeyPatch ) -> None: From 59414fed8454554726de81e0c1819976d788d075 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:40:13 -0400 Subject: [PATCH 17/38] Keep a retracted definition passing on what its body reaches Retracting a definition's statement left its `lean:` declaration in the build, but status gave it no reach, so a proof using it showed as proved while the CI audit flagged it as resting on the open statements the definition's body uses. A retracted definition now passes on the union of its prerequisites' reaches, as a stated one does. It is not open itself: CI rejects a `sorry` in a definition, so it cannot stand for an unproved claim. --- autoform_cli/status.py | 9 +++++---- tests/test_status.py | 25 +++++++++++++++++++++++++ 2 files changed, 30 insertions(+), 4 deletions(-) diff --git a/autoform_cli/status.py b/autoform_cli/status.py index 4d197719..77081565 100644 --- a/autoform_cli/status.py +++ b/autoform_cli/status.py @@ -181,15 +181,16 @@ def done(dependency_id: str, attribute: str) -> bool: can_prove = stated and not unstated and not unmet waiting = unstated + unmet if stated or definition else unstated # Mathlib and unstated nodes reach nothing, so they get no entry, - # except a theorem whose statement a revision retracted while its + # except an article whose statement a revision retracted while its # `lean:` still names the old declaration: that declaration and its - # sorry stay in the build until Formalize restates it. + # sorry stay in the build until Formalize restates it. A retracted + # definition is not open, but its body still reaches what it used. if not node.mathlib: reached = frozenset().union(*(reaches.get(other, frozenset()) for other in node.dependencies)) assumes = tuple(sorted(reached)) - if proved: + if proved or (definition and node.lean): reaches[node_id] = reached - elif stated or (node.lean and not definition): + elif stated or node.lean: reaches[node_id] = frozenset({node_id}).union( *(reaches.get(other, frozenset()) for other in node.statement_dependencies) ) diff --git a/tests/test_status.py b/tests/test_status.py index c269fd63..fb4af63b 100644 --- a/tests/test_status.py +++ b/tests/test_status.py @@ -371,6 +371,31 @@ def test_a_retracted_theorem_still_naming_its_lean_stays_an_open_statement(tmp_p assert (statuses["reduction"].key, statuses["reduction"].assumes) == ("proved", ()) +@pytest.mark.parametrize("policy", ["forbidden", "allowed"]) +def test_a_retracted_definition_still_passes_on_what_its_body_reaches(tmp_path: Path, policy: str) -> None: + """A retracted definition is not open, but its body stays in the build and still uses its open statements.""" + blueprint = tmp_path / "blueprint" + _node(blueprint, "README.md", open_statements=policy) + _node(blueprint, "open.md", declaration="theorem", statement="formalized", lean="Ns.open") + _node(blueprint, "old.md", "## Proof depends on\n\n- [Open](open.md)\n", declaration="def", lean="Ns.old") + _node( + blueprint, + "uses.md", + "## Proof depends on\n\n- [Old](old.md)\n", + declaration="theorem", + statement="formalized", + proof="formalized", + lean="Ns.uses", + ) + + statuses = derive(load_graph(blueprint)) + + if policy == "allowed": + assert (statuses["uses"].key, statuses["uses"].assumes) == ("conditional", ("open",)) + else: + assert (statuses["uses"].key, statuses["uses"].assumes) == ("proved", ()) + + @pytest.mark.parametrize("open_statements", [False, True]) def test_a_missing_dependency_counts_as_neither_stated_nor_proved( tmp_path: Path, open_statements: bool From 12762778ed68109c4b8f2051c6384cbaab4ce992 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:33:37 -0400 Subject: [PATCH 18/38] Keep the open probe's helpers out of a namespace The open-statement probe defined its helpers inside namespace AutoformOpenStatementAudit and read the article table with an unqualified Json.parse. Inside a namespace Lean resolves a name to a constant under that namespace first, so a root module defining AutoformOpenStatementAudit.Json.parse replaced the parser, forged the table and let a sorry theorem recorded as proved pass the audit. The helpers now live at the top level under distinctive autoformOpenAudit* names, where a clashing root constant fails as already declared and one matching a library name makes the reference ambiguous, and the parser is spelled _root_.Lean.Json.parse. Real-Lean tests cover the namespaced hijack and a root-level Json.parse forgery. --- .../templates/github/autoform_audit.py | 47 +++++++++--------- .../.github/autoform_audit.py | 47 +++++++++--------- tests/test_lake_artifact_audit.py | 49 +++++++++++++++++-- 3 files changed, 95 insertions(+), 48 deletions(-) diff --git a/autoform_cli/templates/github/autoform_audit.py b/autoform_cli/templates/github/autoform_audit.py index 0f256c3d..a93289e1 100755 --- a/autoform_cli/templates/github/autoform_audit.py +++ b/autoform_cli/templates/github/autoform_audit.py @@ -459,10 +459,15 @@ def render_open_probe( open Lean Elab Command -namespace AutoformOpenStatementAudit +-- No namespace: inside one, Lean resolves a name to a constant under that +-- namespace before any other, so a root module could define, say, +-- `.Json.parse` and silently replace the parser. At the top level a +-- root constant with a helper's name fails as already declared, and one that +-- matches a library name makes the reference ambiguous; both fail closed. The +-- parser is fully qualified, so a project's own `Json.parse` does not even do that. /-- A name spelled as its components: strings, and numbers for numeric ones. -/ -def nameOf (json : Json) : Except String Name := do +def autoformOpenAuditNameOf (json : Json) : Except String Name := do let mut name := Name.anonymous for part in (← json.getArr?) do match part with @@ -472,18 +477,18 @@ def nameOf (json : Json) : Except String Name := do /-- Per article declaration: its name, its article, whether the article is an open statement, and the open statements the declaration may rest on. -/ -def readArticles (text : String) : Except String (Array (Name × String × Bool × Array Name)) := do - (← (← Json.parse text).getArr?).mapM fun entry => do - let declName ← nameOf (← entry.getArrVal? 0) +def autoformOpenAuditReadArticles (text : String) : Except String (Array (Name × String × Bool × Array Name)) := do + (← (← _root_.Lean.Json.parse text).getArr?).mapM fun entry => do + let declName ← autoformOpenAuditNameOf (← entry.getArrVal? 0) let article ← (← entry.getArrVal? 1).getStr? let isOpen ← (← entry.getArrVal? 2).getBool? - let allowedOpen ← (← (← entry.getArrVal? 3).getArr?).mapM nameOf + let allowedOpen ← (← (← entry.getArrVal? 3).getArr?).mapM autoformOpenAuditNameOf return (declName, article, isOpen, allowedOpen) /-- The types and constructors of one mutual inductive block refer to each other. Nothing else in a safe environment does, so the walk treats a block as one node and needs no cycle handling. -/ -def blockOf (env : Environment) (declName : Name) : Name := +def autoformOpenAuditBlockOf (env : Environment) (declName : Name) : Name := let induct := match env.find? declName with | some (.ctorInfo info) => info.induct | _ => declName @@ -494,7 +499,7 @@ def blockOf (env : Environment) (declName : Name) : Name := /-- The constants a node refers to, as `collectAxioms` follows them, except that an open statement's proof is not entered: a dependent rests on the statement, whatever its proof uses. -/ -def edges (env : Environment) (openSet : Std.HashSet Name) (node : Name) : Array Name := +def autoformOpenAuditEdges (env : Environment) (openSet : Std.HashSet Name) (node : Name) : Array Name := match env.find? node with | some (.inductInfo info) => info.all.foldl (init := (#[] : Array Name)) fun used induct => @@ -513,35 +518,33 @@ def edges (env : Environment) (openSet : Std.HashSet Name) (node : Name) : Array /-- The open statements a node rests on. A constant outside the root package contributes none: the audit requires those to be free of `sorry`. -/ -partial def openHits (env : Environment) (isRoot : Name → Bool) (openSet : Std.HashSet Name) +partial def autoformOpenAuditOpenHits (env : Environment) (isRoot : Name → Bool) (openSet : Std.HashSet Name) (node : Name) : StateM (Std.HashMap Name (Array Name)) (Array Name) := do if let some known := (← get).get? node then return known -- In progress. Only unsafe recursion comes back here, and the audit rejects it. modify (·.insert node #[]) let mut hits : Array Name := if openSet.contains node then #[node] else #[] - for used in edges env openSet node do + for used in autoformOpenAuditEdges env openSet node do if used == ``sorryAx || !isRoot used then continue - let target := blockOf env used + let target := autoformOpenAuditBlockOf env used if target == node then continue - for hit in (← openHits env isRoot openSet target) do + for hit in (← autoformOpenAuditOpenHits env isRoot openSet target) do unless hits.contains hit do hits := hits.push hit hits := hits.qsort Name.lt modify (·.insert node hits) return hits -def nameList (names : Array Name) : MessageData := +def autoformOpenAuditNameList (names : Array Name) : MessageData := MessageData.joinSep (names.toList.map MessageData.ofName) ", " -end AutoformOpenStatementAudit - run_cmd do let targetModules : List Name := [{target_modules}] let allowed : List Name := [``propext, ``Classical.choice, ``Quot.sound] - let articles ← match AutoformOpenStatementAudit.readArticles {articles} with + let articles ← match autoformOpenAuditReadArticles {articles} with | .ok articles => pure articles | .error message => throwError "cannot read the article table: {{message}}" let env ← getEnv @@ -596,8 +599,8 @@ def nameList (names : Array Name) : MessageData := externalSorry := externalSorry.insert used tainted if tainted then errors := errors.push m!"{{declName}} uses {{used}}, which is outside the root package and depends on sorry" - let (hits, cache) := Id.run ((AutoformOpenStatementAudit.openHits env isRoot openSet - (AutoformOpenStatementAudit.blockOf env declName)).run hitCache) + let (hits, cache) := Id.run ((autoformOpenAuditOpenHits env isRoot openSet + (autoformOpenAuditBlockOf env declName)).run hitCache) hitCache := cache if errors.size == reported && usedAxioms.contains ``sorryAx && hits.isEmpty then errors := errors.push m!"{{declName}} depends on sorry outside every declared open statement" @@ -614,12 +617,12 @@ def nameList (names : Array Name) : MessageData := else logInfo m!"sorry-free: {{declName}} [{{article}}]" else - let (hits, cache) := Id.run ((AutoformOpenStatementAudit.openHits env isRoot openSet - (AutoformOpenStatementAudit.blockOf env declName)).run hitCache) + let (hits, cache) := Id.run ((autoformOpenAuditOpenHits env isRoot openSet + (autoformOpenAuditBlockOf env declName)).run hitCache) hitCache := cache let undeclared := hits.filter (fun hit => !allowedOpen.contains hit) unless undeclared.isEmpty do - errors := errors.push m!"{{declName}} [{{article}}] rests on open statement(s) {{AutoformOpenStatementAudit.nameList undeclared}}, which its article's Markdown dependencies do not reach; add the dependency to the article or stop using them" + errors := errors.push m!"{{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList undeclared}}, which its article's Markdown dependencies do not reach; add the dependency to the article or stop using them" if openSet.contains declName then unless isOpen do errors := errors.push m!"{{declName}} [{{article}}] is an open statement, but its article records it as proved" @@ -635,7 +638,7 @@ def nameList (names : Array Name) : MessageData := logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" else if !hits.isEmpty then conditionalCount := conditionalCount + 1 - logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{AutoformOpenStatementAudit.nameList hits}}" + logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList hits}}" else unless (← Lean.collectAxioms declName).contains ``sorryAx do logInfo m!"sorry-free: {{declName}} [{{article}}]" for error in errors do diff --git a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py index 0f256c3d..a93289e1 100755 --- a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py +++ b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py @@ -459,10 +459,15 @@ def render_open_probe( open Lean Elab Command -namespace AutoformOpenStatementAudit +-- No namespace: inside one, Lean resolves a name to a constant under that +-- namespace before any other, so a root module could define, say, +-- `.Json.parse` and silently replace the parser. At the top level a +-- root constant with a helper's name fails as already declared, and one that +-- matches a library name makes the reference ambiguous; both fail closed. The +-- parser is fully qualified, so a project's own `Json.parse` does not even do that. /-- A name spelled as its components: strings, and numbers for numeric ones. -/ -def nameOf (json : Json) : Except String Name := do +def autoformOpenAuditNameOf (json : Json) : Except String Name := do let mut name := Name.anonymous for part in (← json.getArr?) do match part with @@ -472,18 +477,18 @@ def nameOf (json : Json) : Except String Name := do /-- Per article declaration: its name, its article, whether the article is an open statement, and the open statements the declaration may rest on. -/ -def readArticles (text : String) : Except String (Array (Name × String × Bool × Array Name)) := do - (← (← Json.parse text).getArr?).mapM fun entry => do - let declName ← nameOf (← entry.getArrVal? 0) +def autoformOpenAuditReadArticles (text : String) : Except String (Array (Name × String × Bool × Array Name)) := do + (← (← _root_.Lean.Json.parse text).getArr?).mapM fun entry => do + let declName ← autoformOpenAuditNameOf (← entry.getArrVal? 0) let article ← (← entry.getArrVal? 1).getStr? let isOpen ← (← entry.getArrVal? 2).getBool? - let allowedOpen ← (← (← entry.getArrVal? 3).getArr?).mapM nameOf + let allowedOpen ← (← (← entry.getArrVal? 3).getArr?).mapM autoformOpenAuditNameOf return (declName, article, isOpen, allowedOpen) /-- The types and constructors of one mutual inductive block refer to each other. Nothing else in a safe environment does, so the walk treats a block as one node and needs no cycle handling. -/ -def blockOf (env : Environment) (declName : Name) : Name := +def autoformOpenAuditBlockOf (env : Environment) (declName : Name) : Name := let induct := match env.find? declName with | some (.ctorInfo info) => info.induct | _ => declName @@ -494,7 +499,7 @@ def blockOf (env : Environment) (declName : Name) : Name := /-- The constants a node refers to, as `collectAxioms` follows them, except that an open statement's proof is not entered: a dependent rests on the statement, whatever its proof uses. -/ -def edges (env : Environment) (openSet : Std.HashSet Name) (node : Name) : Array Name := +def autoformOpenAuditEdges (env : Environment) (openSet : Std.HashSet Name) (node : Name) : Array Name := match env.find? node with | some (.inductInfo info) => info.all.foldl (init := (#[] : Array Name)) fun used induct => @@ -513,35 +518,33 @@ def edges (env : Environment) (openSet : Std.HashSet Name) (node : Name) : Array /-- The open statements a node rests on. A constant outside the root package contributes none: the audit requires those to be free of `sorry`. -/ -partial def openHits (env : Environment) (isRoot : Name → Bool) (openSet : Std.HashSet Name) +partial def autoformOpenAuditOpenHits (env : Environment) (isRoot : Name → Bool) (openSet : Std.HashSet Name) (node : Name) : StateM (Std.HashMap Name (Array Name)) (Array Name) := do if let some known := (← get).get? node then return known -- In progress. Only unsafe recursion comes back here, and the audit rejects it. modify (·.insert node #[]) let mut hits : Array Name := if openSet.contains node then #[node] else #[] - for used in edges env openSet node do + for used in autoformOpenAuditEdges env openSet node do if used == ``sorryAx || !isRoot used then continue - let target := blockOf env used + let target := autoformOpenAuditBlockOf env used if target == node then continue - for hit in (← openHits env isRoot openSet target) do + for hit in (← autoformOpenAuditOpenHits env isRoot openSet target) do unless hits.contains hit do hits := hits.push hit hits := hits.qsort Name.lt modify (·.insert node hits) return hits -def nameList (names : Array Name) : MessageData := +def autoformOpenAuditNameList (names : Array Name) : MessageData := MessageData.joinSep (names.toList.map MessageData.ofName) ", " -end AutoformOpenStatementAudit - run_cmd do let targetModules : List Name := [{target_modules}] let allowed : List Name := [``propext, ``Classical.choice, ``Quot.sound] - let articles ← match AutoformOpenStatementAudit.readArticles {articles} with + let articles ← match autoformOpenAuditReadArticles {articles} with | .ok articles => pure articles | .error message => throwError "cannot read the article table: {{message}}" let env ← getEnv @@ -596,8 +599,8 @@ def nameList (names : Array Name) : MessageData := externalSorry := externalSorry.insert used tainted if tainted then errors := errors.push m!"{{declName}} uses {{used}}, which is outside the root package and depends on sorry" - let (hits, cache) := Id.run ((AutoformOpenStatementAudit.openHits env isRoot openSet - (AutoformOpenStatementAudit.blockOf env declName)).run hitCache) + let (hits, cache) := Id.run ((autoformOpenAuditOpenHits env isRoot openSet + (autoformOpenAuditBlockOf env declName)).run hitCache) hitCache := cache if errors.size == reported && usedAxioms.contains ``sorryAx && hits.isEmpty then errors := errors.push m!"{{declName}} depends on sorry outside every declared open statement" @@ -614,12 +617,12 @@ def nameList (names : Array Name) : MessageData := else logInfo m!"sorry-free: {{declName}} [{{article}}]" else - let (hits, cache) := Id.run ((AutoformOpenStatementAudit.openHits env isRoot openSet - (AutoformOpenStatementAudit.blockOf env declName)).run hitCache) + let (hits, cache) := Id.run ((autoformOpenAuditOpenHits env isRoot openSet + (autoformOpenAuditBlockOf env declName)).run hitCache) hitCache := cache let undeclared := hits.filter (fun hit => !allowedOpen.contains hit) unless undeclared.isEmpty do - errors := errors.push m!"{{declName}} [{{article}}] rests on open statement(s) {{AutoformOpenStatementAudit.nameList undeclared}}, which its article's Markdown dependencies do not reach; add the dependency to the article or stop using them" + errors := errors.push m!"{{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList undeclared}}, which its article's Markdown dependencies do not reach; add the dependency to the article or stop using them" if openSet.contains declName then unless isOpen do errors := errors.push m!"{{declName}} [{{article}}] is an open statement, but its article records it as proved" @@ -635,7 +638,7 @@ def nameList (names : Array Name) : MessageData := logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" else if !hits.isEmpty then conditionalCount := conditionalCount + 1 - logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{AutoformOpenStatementAudit.nameList hits}}" + logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList hits}}" else unless (← Lean.collectAxioms declName).contains ``sorryAx do logInfo m!"sorry-free: {{declName}} [{{article}}]" for error in errors do diff --git a/tests/test_lake_artifact_audit.py b/tests/test_lake_artifact_audit.py index e1ef465c..47b61989 100644 --- a/tests/test_lake_artifact_audit.py +++ b/tests/test_lake_artifact_audit.py @@ -589,8 +589,8 @@ def test_open_probe_spells_names_as_components(helper: ModuleType, tmp_path: Pat probe = helper.render_open_probe(("Fixture",), helper.load_assumption_contract(contract)) - table = next(line for line in probe.splitlines() if "readArticles " in line and " with" in line) - literal = table.split("readArticles ", 1)[1].rsplit(" with", 1)[0] + table = next(line for line in probe.splitlines() if "ReadArticles " in line and " with" in line) + literal = table.split("ReadArticles ", 1)[1].rsplit(" with", 1)[0] assert json.loads(json.loads(literal)) == [ [["Fixture", "a.b c"], "open", True, [["Fixture", "a.b c"]]], [["Fixture", "x", 1], 'odd "id" \\', False, [["Fixture", "a.b c"]]], @@ -631,7 +631,7 @@ def run(*arguments: str) -> subprocess.CompletedProcess[str]: assert prepared.stdout == ( "prepared open-statement audit for 1 root-package module(s) and 1 open statement candidate(s)\n" ) - assert "AutoformOpenStatementAudit.readArticles" in probe.read_text(encoding="utf-8") + assert "autoformOpenAuditReadArticles" in probe.read_text(encoding="utf-8") _contract_file(contract, _contract(open_statements=False)) forbidden = run("--open-statements", str(contract), "Fixture", str(archive), str(probe)) @@ -683,6 +683,23 @@ def run(*arguments: str) -> subprocess.CompletedProcess[str]: end Fixture """ +# A sorry theorem that would pass as open if the probe read this forged table. +_FORGED_TABLE = json.dumps(json.dumps([[["Fixture", "cheat"], "x", True, [["Fixture", "cheat"]]]])) + +_FORGED_LEAN = """import Lean.Data.Json + +namespace Fixture + +theorem cheat : 2 + 2 = 5 := sorry + +theorem fully : 2 + 2 = 5 := cheat + +end Fixture + +def {parser} (_ : String) : Except String Lean.Json := + Lean.Json.parse {table} +""" + @pytest.fixture(scope="module") def open_projects(tmp_path_factory: pytest.TempPathFactory) -> dict[str, tuple[Path, Path]]: @@ -694,7 +711,14 @@ def open_projects(tmp_path_factory: pytest.TempPathFactory) -> dict[str, tuple[P _write(dependency / "lakefile.toml", 'name = "Dep"\ndefaultTargets = ["Dep"]\n\n[[lean_lib]]\nname = "Dep"\n') _write(dependency / "Dep.lean", _DEPENDENCY_LEAN) projects: dict[str, tuple[Path, Path]] = {} - for name, source in (("open", _OPEN_LEAN), ("bad", _BAD_LEAN)): + for name, source in ( + ("open", _OPEN_LEAN), + ("bad", _BAD_LEAN), + # The probe's helpers once lived in this namespace, where the name took + # precedence over the library parser. + ("hijack", _FORGED_LEAN.format(parser="AutoformOpenStatementAudit.Json.parse", table=_FORGED_TABLE)), + ("forged", _FORGED_LEAN.format(parser="Json.parse", table=_FORGED_TABLE)), + ): project = root / name _write(project / "lean-toolchain", "leanprover/lean4:v4.32.2\n") _write( @@ -811,6 +835,23 @@ def test_open_probe_rejects_sorry_outside_an_open_statement_body( assert message in output +@pytest.mark.parametrize("project", ["hijack", "forged"]) +def test_open_probe_reads_its_table_with_the_library_parser( + helper: ModuleType, open_projects: dict[str, tuple[Path, Path]], project: str +) -> None: + audited = _audit( + helper, + open_projects[project], + _contract(_article("fully", ["Fixture.fully"]), _article("cheat", ["Fixture.cheat"])), + ) + + output = audited.stdout + audited.stderr + assert audited.returncode != 0, output + assert "Fixture.cheat contains sorry but is not an open statement" in output + assert "Fixture.fully depends on sorry outside every declared open statement" in output + assert "kernel trust clean" not in output + + def test_strict_probe_still_rejects_every_sorry( helper: ModuleType, open_projects: dict[str, tuple[Path, Path]] ) -> None: From a7b8672e412e468e481492f4d454738c8ea3046e Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:33:37 -0400 Subject: [PATCH 19/38] Log no clean result for a declaration that failed the open audit The open probe's article loop printed sorry-free, conditional and sorry-free-open-statement info lines even when the same declaration had an error, from the roots loop (an unexpected axiom) or from the article loop itself (an open statement its Markdown does not reach), so the log showed a clean line right next to the error. Those lines are now logged only for a declaration that added no error; the verdict is unchanged. --- .../templates/github/autoform_audit.py | 13 ++++-- .../.github/autoform_audit.py | 13 ++++-- tests/test_lake_artifact_audit.py | 41 +++++++++++++++++++ 3 files changed, 61 insertions(+), 6 deletions(-) diff --git a/autoform_cli/templates/github/autoform_audit.py b/autoform_cli/templates/github/autoform_audit.py index a93289e1..21b3acf1 100755 --- a/autoform_cli/templates/github/autoform_audit.py +++ b/autoform_cli/templates/github/autoform_audit.py @@ -565,6 +565,8 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := let mut errors : Array MessageData := #[] let mut hitCache : Std.HashMap Name (Array Name) := {{}} let mut externalSorry : Std.HashMap Name Bool := {{}} + -- Declarations with an error get no info line that reads as a clean result. + let mut failed : Std.HashSet Name := {{}} for declName in roots do let some info := env.find? declName | continue let reported := errors.size @@ -604,9 +606,12 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := hitCache := cache if errors.size == reported && usedAxioms.contains ``sorryAx && hits.isEmpty then errors := errors.push m!"{{declName}} depends on sorry outside every declared open statement" + if errors.size != reported then + failed := failed.insert declName let mut openCount : Nat := 0 let mut conditionalCount : Nat := 0 for (declName, article, isOpen, allowedOpen) in articles do + let reported := errors.size match env.find? declName with | none => errors := errors.push m!"{{declName}} [{{article}}] is not a declaration of the Lean build; fix the article's lean: name or build the module that declares it" @@ -634,13 +639,15 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := logInfo m!"open statement (proof is sorry): {{declName}} [{{article}}]" else if (← Lean.collectAxioms declName).contains ``sorryAx then logInfo m!"open statement (proof depends on sorry elsewhere): {{declName}} [{{article}}]" - else + else if errors.size == reported && !failed.contains declName then logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" else if !hits.isEmpty then conditionalCount := conditionalCount + 1 - logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList hits}}" + if errors.size == reported && !failed.contains declName then + logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList hits}}" else unless (← Lean.collectAxioms declName).contains ``sorryAx do - logInfo m!"sorry-free: {{declName}} [{{article}}]" + if errors.size == reported && !failed.contains declName then + logInfo m!"sorry-free: {{declName}} [{{article}}]" for error in errors do logError error if roots.isEmpty then diff --git a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py index a93289e1..21b3acf1 100755 --- a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py +++ b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py @@ -565,6 +565,8 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := let mut errors : Array MessageData := #[] let mut hitCache : Std.HashMap Name (Array Name) := {{}} let mut externalSorry : Std.HashMap Name Bool := {{}} + -- Declarations with an error get no info line that reads as a clean result. + let mut failed : Std.HashSet Name := {{}} for declName in roots do let some info := env.find? declName | continue let reported := errors.size @@ -604,9 +606,12 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := hitCache := cache if errors.size == reported && usedAxioms.contains ``sorryAx && hits.isEmpty then errors := errors.push m!"{{declName}} depends on sorry outside every declared open statement" + if errors.size != reported then + failed := failed.insert declName let mut openCount : Nat := 0 let mut conditionalCount : Nat := 0 for (declName, article, isOpen, allowedOpen) in articles do + let reported := errors.size match env.find? declName with | none => errors := errors.push m!"{{declName}} [{{article}}] is not a declaration of the Lean build; fix the article's lean: name or build the module that declares it" @@ -634,13 +639,15 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := logInfo m!"open statement (proof is sorry): {{declName}} [{{article}}]" else if (← Lean.collectAxioms declName).contains ``sorryAx then logInfo m!"open statement (proof depends on sorry elsewhere): {{declName}} [{{article}}]" - else + else if errors.size == reported && !failed.contains declName then logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" else if !hits.isEmpty then conditionalCount := conditionalCount + 1 - logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList hits}}" + if errors.size == reported && !failed.contains declName then + logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList hits}}" else unless (← Lean.collectAxioms declName).contains ``sorryAx do - logInfo m!"sorry-free: {{declName}} [{{article}}]" + if errors.size == reported && !failed.contains declName then + logInfo m!"sorry-free: {{declName}} [{{article}}]" for error in errors do logError error if roots.isEmpty then diff --git a/tests/test_lake_artifact_audit.py b/tests/test_lake_artifact_audit.py index 47b61989..7dc85ed7 100644 --- a/tests/test_lake_artifact_audit.py +++ b/tests/test_lake_artifact_audit.py @@ -680,6 +680,14 @@ def run(*arguments: str) -> subprocess.CompletedProcess[str]: theorem uses_dependency_sorry : True := Dep.dep_sorry +theorem native : 10 + 10 = 20 := by native_decide + +theorem uses_native : 10 + 10 = 20 := native + +theorem native_reduction : 1 + 1 = 2 := by have := native; exact open_stmt + +theorem native_open : 3 + 3 = 6 := by native_decide + end Fixture """ @@ -835,6 +843,39 @@ def test_open_probe_rejects_sorry_outside_an_open_statement_body( assert message in output +def test_open_probe_logs_no_clean_result_for_a_failing_declaration( + helper: ModuleType, open_projects: dict[str, tuple[Path, Path]] +) -> None: + audited = _audit( + helper, + open_projects["bad"], + _contract( + _OPEN_ARTICLE, + _article("reduction", ["Fixture.reduction"]), + _article("uses-native", ["Fixture.uses_native"]), + _article("native-reduction", ["Fixture.native_reduction"], allowed=["Fixture.open_stmt"]), + _article("native-open", ["Fixture.native_open"], is_open=True, allowed=["Fixture.native_open"]), + _article("clean", ["Fixture.clean"]), + ), + ) + + output = audited.stdout + audited.stderr + assert audited.returncode != 0, output + for message in ( + "Fixture.reduction [reduction] rests on open statement(s) Fixture.open_stmt, which", + "Fixture.uses_native depends on unexpected axiom", + "Fixture.native_reduction depends on unexpected axiom", + "Fixture.native_open depends on unexpected axiom", + "open statement (proof is sorry): Fixture.open_stmt [open]", + "sorry-free: Fixture.clean [clean]", + ): + assert message in output + for name in ("reduction", "uses_native", "native_reduction", "native_open"): + for line in output.splitlines(): + if f"Fixture.{name} [" in line: + assert not line.startswith(("sorry-free:", "conditional:", "open statement (")), line + + @pytest.mark.parametrize("project", ["hijack", "forged"]) def test_open_probe_reads_its_table_with_the_library_parser( helper: ModuleType, open_projects: dict[str, tuple[Path, Path]], project: str From 5abb13394691ebbc82a403dedec04030f6036bb4 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:33:37 -0400 Subject: [PATCH 20/38] Explain recursive auxiliaries in the open audit's sorry error Lean compiles a mutually or well-founded recursive proof into auxiliaries such as X._f and X._unary, so a sorry written directly in an open theorem's recursive body is reported on the auxiliary, and the error told the user to do what they had already done. The audit still rejects it; the message now says that a recursive open statement's proof must be exactly sorry. --- autoform_cli/templates/github/autoform_audit.py | 2 +- .../.github/autoform_audit.py | 2 +- tests/test_lake_artifact_audit.py | 14 ++++++++++++++ 3 files changed, 16 insertions(+), 2 deletions(-) diff --git a/autoform_cli/templates/github/autoform_audit.py b/autoform_cli/templates/github/autoform_audit.py index 21b3acf1..2d19a210 100755 --- a/autoform_cli/templates/github/autoform_audit.py +++ b/autoform_cli/templates/github/autoform_audit.py @@ -587,7 +587,7 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := if typeConstants.contains ``sorryAx then errors := errors.push m!"{{declName}} has sorry in its statement; state the claim in full and keep sorry only in the proof" else if valueConstants.contains ``sorryAx && !openSet.contains declName then - errors := errors.push m!"{{declName}} contains sorry but is not an open statement: only a theorem that an open article's lean: names may keep a sorry, written directly in its own proof, not in a helper, where clause or definition" + errors := errors.push m!"{{declName}} contains sorry but is not an open statement: only a theorem that an open article's lean: names may keep a sorry, written directly in its own proof, not in a helper, where clause or definition; Lean compiles a recursive proof into auxiliaries such as _f and _unary, so a recursive open statement's proof must be exactly sorry" let mut external : Array Name := #[] for used in typeConstants ++ valueConstants do if used == ``sorryAx || isRoot used || external.contains used then diff --git a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py index 21b3acf1..2d19a210 100755 --- a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py +++ b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py @@ -587,7 +587,7 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := if typeConstants.contains ``sorryAx then errors := errors.push m!"{{declName}} has sorry in its statement; state the claim in full and keep sorry only in the proof" else if valueConstants.contains ``sorryAx && !openSet.contains declName then - errors := errors.push m!"{{declName}} contains sorry but is not an open statement: only a theorem that an open article's lean: names may keep a sorry, written directly in its own proof, not in a helper, where clause or definition" + errors := errors.push m!"{{declName}} contains sorry but is not an open statement: only a theorem that an open article's lean: names may keep a sorry, written directly in its own proof, not in a helper, where clause or definition; Lean compiles a recursive proof into auxiliaries such as _f and _unary, so a recursive open statement's proof must be exactly sorry" let mut external : Array Name := #[] for used in typeConstants ++ valueConstants do if used == ``sorryAx || isRoot used || external.contains used then diff --git a/tests/test_lake_artifact_audit.py b/tests/test_lake_artifact_audit.py index 7dc85ed7..91a487a7 100644 --- a/tests/test_lake_artifact_audit.py +++ b/tests/test_lake_artifact_audit.py @@ -680,6 +680,15 @@ def run(*arguments: str) -> subprocess.CompletedProcess[str]: theorem uses_dependency_sorry : True := Dep.dep_sorry +mutual +theorem evenA : ∀ n : Nat, n + 0 = n + | 0 => rfl + | n + 1 => by have := oddB n; sorry +theorem oddB : ∀ n : Nat, 0 + n = n + | 0 => rfl + | n + 1 => by have := evenA n; omega +end + theorem native : 10 + 10 = 20 := by native_decide theorem uses_native : 10 + 10 = 20 := native @@ -827,6 +836,7 @@ def test_open_probe_rejects_sorry_outside_an_open_statement_body( _article("where", ["Fixture.where_sorry"], is_open=True, allowed=["Fixture.where_sorry"]), _article("type", ["Fixture.type_sorry"], is_open=True, allowed=["Fixture.type_sorry"]), _article("missing", ["Fixture.missing"]), + _article("even", ["Fixture.evenA"], is_open=True, allowed=["Fixture.evenA"]), ), ) @@ -834,6 +844,10 @@ def test_open_probe_rejects_sorry_outside_an_open_statement_body( assert audited.returncode != 0, output for message in ( "Fixture.helper_sorry contains sorry but is not an open statement", + "Fixture.evenA._f contains sorry but is not an open statement: only a theorem that an open " + "article's lean: names may keep a sorry, written directly in its own proof, not in a helper, " + "where clause or definition; Lean compiles a recursive proof into auxiliaries such as _f and " + "_unary, so a recursive open statement's proof must be exactly sorry", "Fixture.where_sorry.aux contains sorry but is not an open statement", "Fixture.type_sorry has sorry in its statement", "Fixture.uses_dependency_sorry uses Dep.dep_sorry, which is outside the root package and depends on sorry", From e9edaf10479f795f18e857c658272f56b05754c6 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:38:26 -0400 Subject: [PATCH 21/38] Name the probe in Lean build freshness messages run_probe takes a label, but its missing-manifest and stale-artifact messages always said to run `lake build` "before extracting skeletons", so `work impact` on a stale build talked about skeletons. Both messages now say "before running the impact probe" for the impact probe, while the skeleton probe keeps its exact wording. --- autoform_cli/skeleton.py | 16 ++++++++++++---- tests/test_impact.py | 22 ++++++++++++++++++++++ 2 files changed, 34 insertions(+), 4 deletions(-) diff --git a/autoform_cli/skeleton.py b/autoform_cli/skeleton.py index 9b701448..1a62e439 100644 --- a/autoform_cli/skeleton.py +++ b/autoform_cli/skeleton.py @@ -1364,8 +1364,14 @@ def _probe_modules(probe: str) -> tuple[str, ...]: return tuple(dict.fromkeys(modules)) +def _probe_purpose(label: str) -> str: + """What `lake build` must come before, as the freshness messages say it.""" + + return "extracting skeletons" if label == "skeleton probe" else f"running the {label}" + + def _check_artifacts_fresh( - lake: str, lean_root: Path, modules: tuple[str, ...], *, timeout: float + lake: str, lean_root: Path, modules: tuple[str, ...], *, timeout: float, label: str = "skeleton probe" ) -> None: """Ask Lake to prove that imported artifacts match their exact inputs. @@ -1382,7 +1388,9 @@ def _check_artifacts_fresh( ) detail = (result.stderr or result.stdout).strip() if result.returncode == _LAKE_NO_BUILD_EXIT: - raise SkeletonError([f"Lean build artifacts are stale; run `lake build` before extracting skeletons\n{detail}"]) + raise SkeletonError( + [f"Lean build artifacts are stale; run `lake build` before {_probe_purpose(label)}\n{detail}"] + ) if result.returncode != 0: raise SkeletonError( [ @@ -1413,10 +1421,10 @@ def run_probe( raise SkeletonError(["lake is not on PATH; a built Lean project is required to extract skeletons"]) if not (lean_root / "lake-manifest.json").is_file(): raise SkeletonError( - ["lake-manifest.json is missing; run `lake build` before extracting skeletons"] + [f"lake-manifest.json is missing; run `lake build` before {_probe_purpose(label)}"] ) modules = _probe_modules(probe) - _check_artifacts_fresh(lake, lean_root, modules, timeout=freshness_timeout) + _check_artifacts_fresh(lake, lean_root, modules, timeout=freshness_timeout, label=label) deadline = time.monotonic() + timeout env = os.environ.copy() env.pop("PYTHONPATH", None) diff --git a/tests/test_impact.py b/tests/test_impact.py index 374f4816..4996a405 100644 --- a/tests/test_impact.py +++ b/tests/test_impact.py @@ -436,6 +436,28 @@ def test_probe_failures_name_the_impact_probe( assert default.value.issues == (message.format(label="skeleton probe"),) +def test_impact_probe_freshness_messages_never_mention_skeletons(tmp_path: Path, monkeypatch) -> None: + monkeypatch.setattr("autoform_cli.skeleton.shutil.which", lambda executable: "/bin/lake") + monkeypatch.setattr( + "autoform_cli.skeleton._run_bounded_command", + lambda command, **kwargs: subprocess.CompletedProcess(command, 3, stdout="", stderr="Demo is out of date"), + ) + probe = render_impact_probe(imports=["Demo"], project_roots=["Demo"]) + + def issues(label: str | None) -> tuple[str, ...]: + with pytest.raises(SkeletonError) as caught: + skeleton.run_probe(probe, tmp_path, **({} if label is None else {"label": label})) + return caught.value.issues + + missing = "lake-manifest.json is missing; run `lake build` before" + assert issues("impact probe") == (f"{missing} running the impact probe",) + assert issues(None) == (f"{missing} extracting skeletons",) + (tmp_path / "lake-manifest.json").write_text("{}\n", encoding="utf-8") + stale = "Lean build artifacts are stale; run `lake build` before" + assert issues("impact probe") == (f"{stale} running the impact probe\nDemo is out of date",) + assert issues(None) == (f"{stale} extracting skeletons\nDemo is out of date",) + + # --------------------------------------------------------------------------- # # Project modules # --------------------------------------------------------------------------- # From c582ec0c86acafce71159266d2d973a93b9a40c7 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:38:37 -0400 Subject: [PATCH 22/38] Resolve private names, follow aliases, and tighten deprecation in work impact Articles and --declaration name a private declaration as its source does, but probe records carry kernel names, so an article naming one dropped out of the report and `work impact` on it failed. The probe now emits each private constant's user-facing name; a name resolves to its exact record, else to the one private record with that user-facing name, and a name several private records share is refused. Helpers of a private declaration find their owner through it, and the source locator uses the same field instead of stripping the private prefix. A theorem whose value is exactly a constant, as Batteries' `alias` makes, copies its target's type without mentioning the target, so it landed in the proof set and its users went unreported. The probe now emits that constant as `alias_of`, and it counts as a meaning edge. A deprecated constant's own generated companions (`eq_1`, `_simp_1`) no longer count as its users, though real users of a companion still do. A deprecated constant that another article's lean: names lists that article and stays out of deprecated_unused. Helpers the revised article owns, such as a structure's generated constructor and recursor, are repaired under its claim and no longer make a revision non-contained. --- autoform_cli/impact.py | 126 ++++++++++-- autoform_cli/probes/impact_probe.lean | 11 +- tests/test_impact.py | 285 +++++++++++++++++++++++++- 3 files changed, 392 insertions(+), 30 deletions(-) diff --git a/autoform_cli/impact.py b/autoform_cli/impact.py index 132703e2..aefbc355 100644 --- a/autoform_cli/impact.py +++ b/autoform_cli/impact.py @@ -54,6 +54,10 @@ class ConstantRecord: ``internal`` is ``Name.isInternalDetail`` of the user-facing name, so a private declaration someone wrote is not internal, while the companions Lean generates (``_proof_1``, ``match_1``, ``_simp_1``) are. + ``user_name`` is the user-facing name of a private constant, which is how + articles and ``--declaration`` name it, and ``None`` for any other. + ``alias_of`` is the local constant a theorem's value is exactly, as for + Batteries' ``alias``, whose type is copied from that constant's. """ name: str @@ -68,15 +72,22 @@ class ConstantRecord: replacement: str | None = None uses_deprecated: tuple[str, ...] = () value_missing: bool = False + user_name: str | None = None + alias_of: str | None = None @property def meaning_uses(self) -> tuple[str, ...]: - """The local constants this constant's meaning rests on.""" + """The local constants this constant's meaning rests on. - return self.type_uses + self.value_uses if self.kind in _VALUE_IS_MEANING else self.type_uses + An alias's statement is its target's, so it changes with the target's. + """ + + uses = self.type_uses + self.value_uses if self.kind in _VALUE_IS_MEANING else self.type_uses + return (*uses, self.alias_of) if self.alias_of is not None else uses _RECORD_FIELDS: dict[str, tuple[type, ...]] = { + "alias_of": (str, type(None)), "deprecated": (bool,), "instance": (bool,), "internal": (bool,), @@ -86,6 +97,7 @@ def meaning_uses(self) -> tuple[str, ...]: "parent": (str, type(None)), "replacement": (str, type(None)), "type_uses": (list,), + "user_name": (str, type(None)), "uses_deprecated": (list,), "value_missing": (bool,), "value_uses": (list,), @@ -133,7 +145,13 @@ def _constant_record(payload: object) -> ConstantRecord: isinstance(value, list) and not all(isinstance(item, str) for item in value) ): raise SkeletonError([f"the impact probe emitted a malformed {field!r} field"]) - if not payload["name"] or not payload["module"] or payload["kind"] not in _KINDS: + if ( + not payload["name"] + or not payload["module"] + or payload["kind"] not in _KINDS + or payload["user_name"] == "" + or payload["alias_of"] not in (None, *payload["value_uses"]) + ): raise SkeletonError([f"the impact probe emitted a malformed record for {payload['name']!r}"]) return ConstantRecord( name=payload["name"], @@ -148,6 +166,8 @@ def _constant_record(payload: object) -> ConstantRecord: replacement=payload["replacement"], uses_deprecated=tuple(payload["uses_deprecated"]), value_missing=payload["value_missing"], + user_name=payload["user_name"], + alias_of=payload["alias_of"], ) @@ -330,14 +350,26 @@ def as_dict(self) -> dict[str, object]: @dataclass(frozen=True, slots=True) class DeprecatedConstant: - """A local deprecated constant, its replacement, and the local constants that use it.""" + """A local deprecated constant, its replacement, and what still refers to it. + + ``users`` are the local constants that use it; ``articles`` are the + articles other than the revised one whose ``lean:`` names it. The revised + article does not count, since it drops the name in the commit that deletes + the constant. + """ name: str replacement: str | None users: tuple[str, ...] + articles: tuple[str, ...] def as_dict(self) -> dict[str, object]: - return {"name": self.name, "replacement": self.replacement, "users": list(self.users)} + return { + "name": self.name, + "replacement": self.replacement, + "users": list(self.users), + "articles": list(self.articles), + } @dataclass(frozen=True, slots=True) @@ -357,9 +389,18 @@ class ImpactReport: @property def contained(self) -> bool: - """Whether nothing outside the revised article is impacted.""" + """Whether nothing outside the revised article is impacted. - return not (self.statement_impacted or self.proof_impacted or self.helpers) + A helper the revised article owns, such as a structure's generated + constructor or recursor, is repaired under that article's claim, so it + does not count; any other helper, owned or not, does. + """ + + return not ( + self.statement_impacted + or self.proof_impacted + or any(helper.owner != self.article.id for helper in self.helpers) + ) def as_dict(self) -> dict[str, object]: return { @@ -406,14 +447,16 @@ def compute_impact( directly or through internal-detail constants, such as the ``_simp_1`` companion ``simp`` uses or the ``_proof_1`` a definition's nested proof becomes. Definitions belong in P too: their nested proofs can stop - elaborating although their meaning is unchanged. + elaborating although their meaning is unchanged. A theorem whose value is + exactly another constant, as Batteries' ``alias`` makes, has that + constant's statement, so it joins M with it. """ - by_key = {_name_key(name): name for name in records} + resolve = _resolver(records) revised_names: dict[str, str] = {} missing: list[str] = [] for name in declarations: - record_name = by_key.get(_name_key(name)) + record_name = resolve(name) if record_name is None: missing.append(name) else: @@ -445,7 +488,7 @@ def compute_impact( statement_impacted: list[ImpactedArticle] = [] proof_impacted: list[ImpactedArticle] = [] for article in articles: - resolved = [(name, by_key.get(_name_key(name))) for name in article.declarations] + resolved = [(name, resolve(name, f"{article.id}: lean: ")) for name in article.declarations] for _, record_name in resolved: if record_name is not None: named.setdefault(record_name, []).append(article.id) @@ -484,6 +527,7 @@ def compute_impact( record.name, record.replacement, _users_through_internal(record.name, deprecated_users.get(record.name, set()), users, records), + tuple(sorted(set(named.get(record.name, ())) - {revised.id})), ) for record in sorted(records.values(), key=lambda item: item.name) if record.deprecated @@ -502,11 +546,37 @@ def compute_impact( helpers=tuple(helpers), undeclared_dependencies=tuple(undeclared), deprecated=deprecated, - deprecated_unused=tuple(item.name for item in deprecated if not item.users), + deprecated_unused=tuple(item.name for item in deprecated if not item.users and not item.articles), claim_targets=claim_targets, ) +def _resolver(records: Mapping[str, ConstantRecord]) -> Callable[..., str | None]: + """Resolve a user-facing name to a record: its exact name, else the private constant it names. + + Articles and ``--declaration`` name a private constant as its source does, + without the ``_private`` prefix the probe's names carry. A name several + private constants share is refused rather than guessed. + """ + + exact = {_name_key(name): name for name in records} + private: dict[tuple[object, ...], list[str]] = {} + for record in records.values(): + if record.user_name is not None: + private.setdefault(_name_key(record.user_name), []).append(record.name) + + def resolve(name: str, context: str = "") -> str | None: + key = _name_key(name) + if key in exact: + return exact[key] + candidates = sorted(private.get(key, ())) + if len(candidates) > 1: + raise ImpactError(f"{context}{name} names several private declarations: {', '.join(candidates)}") + return candidates[0] if candidates else None + + return resolve + + def _reverse_closure( start: Iterable[str], users: Mapping[str, set[str]], @@ -539,6 +609,20 @@ def _owner(record: ConstantRecord, records: Mapping[str, ConstantRecord], named: return None +def _descends_from(record: ConstantRecord, ancestor: str, records: Mapping[str, ConstantRecord]) -> bool: + """Whether ``ancestor`` is on the ``parent`` chain of ``record``.""" + + seen: set[str] = set() + parent = record.parent + while parent is not None and parent not in seen: + if parent == ancestor: + return True + seen.add(parent) + ancestor_record = records.get(parent) + parent = ancestor_record.parent if ancestor_record is not None else None + return False + + def _reaches(start: str, target: str, articles: Mapping[str, ImpactArticle]) -> bool: """Whether ``start`` reaches ``target`` through Markdown dependencies.""" @@ -564,7 +648,9 @@ def _users_through_internal( """The local users of ``name``; an internal-detail user stands for its own users. An internal-detail user nothing else uses is kept, so that a constant is - never reported unused while something still mentions it. + never reported unused while something still mentions it, unless it is a + companion of ``name`` itself, such as its ``eq_1`` or ``_simp_1``, which + goes when ``name`` does. """ found: set[str] = set() @@ -576,7 +662,7 @@ def _users_through_internal( continue seen.add(user) further = users.get(user, set()) - {user} - if records[user].internal and further: + if records[user].internal and (further or _descends_from(records[user], name, records)): work.extend(sorted(further)) else: found.add(user) @@ -653,7 +739,7 @@ def locate(record: ConstantRecord) -> tuple[str | None, int | None]: if index is None: index = index_project(root) module_path = path_of(record.module, libraries, root) - declaration = index.find(record.name.removeprefix(f"_private.{record.module}.0.")) + declaration = index.find(record.user_name or record.name) if declaration is not None and module_path in (None, declaration.path.as_posix()): return declaration.path.as_posix(), declaration.line return module_path, None @@ -670,7 +756,9 @@ def format_impact(report: ImpactReport) -> list[str]: f"Graph source revision: {report.source_revision}", ] if report.contained: - lines.append(f"Contained: no other article or helper uses {names}, so it can be revised in place.") + lines.append( + f"Contained: nothing outside {report.article.id} uses {names}, so it can be revised in place." + ) for title, impacted in ( ("Statement impacted", report.statement_impacted), ("Proof impacted", report.proof_impacted), @@ -695,8 +783,10 @@ def format_impact(report: ImpactReport) -> list[str]: lines.append("Deprecated:") for item in report.deprecated: replacement = f" -> {item.replacement}" if item.replacement else "" - users = f"used by {', '.join(item.users)}" if item.users else "no users, safe to delete" - lines.append(f" {item.name}{replacement}: {users}") + uses = [f"used by {', '.join(item.users)}"] if item.users else [] + if item.articles: + uses.append(f"named by {', '.join(item.articles)}") + lines.append(f" {item.name}{replacement}: {'; '.join(uses) or 'no users, safe to delete'}") lines.append("Claim targets: " + ", ".join(report.claim_targets)) return lines diff --git a/autoform_cli/probes/impact_probe.lean b/autoform_cli/probes/impact_probe.lean index c86f9da3..a43e59c2 100644 --- a/autoform_cli/probes/impact_probe.lean +++ b/autoform_cli/probes/impact_probe.lean @@ -57,7 +57,11 @@ Theorem values are read too (`allowOpaque`), since proofs break when the statements they use change; an inductive's constructors stand in for a value. `internal` is judged on the user-facing name: a private declaration someone wrote is not an internal detail, while the companions Lean generates for it -(`_proof_1`, `match_1`, `_simp_1`) still are. -/ +(`_proof_1`, `match_1`, `_simp_1`) still are. `user_name` is that name for a +private constant, which is how articles and `--declaration` name it. +`alias_of` is the local constant a theorem's value is exactly, as for +Batteries' `alias`: such a theorem copies its target's type, so its statement +changes with the target's although its type never mentions it. -/ def record (env : Environment) (isLocal : Name → Bool) (c : Name) (info : ConstantInfo) (module : Name) : Json := let value := info.value? (allowOpaque := true) @@ -68,6 +72,9 @@ def record (env : Environment) (isLocal : Name → Bool) (c : Name) (info : Cons let valueMissing := match info with | .thmInfo _ | .defnInfo _ | .opaqueInfo _ => value.isNone | _ => false + let aliasOf := match info, value.map (·.consumeMData) with + | .thmInfo _, some (.const target _) => if isLocal target then some target else none + | _, _ => none let usesDeprecated := (typeConstants ++ valueConstants).foldl (fun acc d => if Linter.isDeprecated env d && !acc.contains d then acc.push d else acc) #[] Json.mkObj [ @@ -75,10 +82,12 @@ def record (env : Environment) (isLocal : Name → Bool) (c : Name) (info : Cons ("kind", Json.str (kindOf info)), ("instance", Json.bool (isInstanceCore env c)), ("internal", Json.bool (privateToUserName c).isInternalDetail), + ("user_name", if isPrivateName c then nameJson (privateToUserName c) else Json.null), ("module", nameJson module), ("parent", ((parentOf env c).map nameJson).getD Json.null), ("type_uses", namesJson (typeConstants.filter isLocal)), ("value_uses", namesJson (valueConstants.filter isLocal)), + ("alias_of", (aliasOf.map nameJson).getD Json.null), ("deprecated", Json.bool (Linter.isDeprecated env c)), ("replacement", ((Linter.getDeprecatedNewName env c).map nameJson).getD Json.null), ("uses_deprecated", namesJson usesDeprecated), diff --git a/tests/test_impact.py b/tests/test_impact.py index 4996a405..82d6c106 100644 --- a/tests/test_impact.py +++ b/tests/test_impact.py @@ -296,9 +296,9 @@ def test_deprecated_constants_report_users_through_internal_details() -> None: # An internal user stands for its own users, unless nothing uses it: a # constant something still mentions is never reported as safe to delete. assert [item.as_dict() for item in report.deprecated] == [ - {"name": "A.old", "replacement": "A.new", "users": ["A.user", "A.wrapped"]}, - {"name": "A.older", "replacement": None, "users": ["A.stray._proof_1"]}, - {"name": "A.unused", "replacement": "A.new", "users": []}, + {"name": "A.old", "replacement": "A.new", "users": ["A.user", "A.wrapped"], "articles": []}, + {"name": "A.older", "replacement": None, "users": ["A.stray._proof_1"], "articles": []}, + {"name": "A.unused", "replacement": "A.new", "users": [], "articles": []}, ] assert report.deprecated_unused == ("A.unused",) assert report.contained @@ -316,11 +316,177 @@ def test_a_contained_revision_says_it_can_be_revised_in_place() -> None: assert format_impact(report) == [ "Revising A.leaf of chapter/leaf [af_leaf]", "Graph source revision: rev", - "Contained: no other article or helper uses A.leaf, so it can be revised in place.", + "Contained: nothing outside chapter/leaf uses A.leaf, so it can be revised in place.", "Claim targets: af_leaf", ] +def test_private_declarations_resolve_by_their_user_facing_names() -> None: + privT = "_private.Demo.B.0.A.privT" + records = _records( + _rec("A.base", "def"), + _rec(privT, module="Demo.B", type_uses=("A.base",), user_name="A.privT"), + _rec(f"{privT}.aux", module="Demo.B", type_uses=("A.base",), user_name="A.privT.aux", parent=privT), + _rec("_private.Demo.C.0.A.privU", "def", module="Demo.C", user_name="A.privU"), + _rec("A.privU", "def", type_uses=("A.base",)), + ) + articles = [ + _article("base", "A.base"), + _article("priv", "A.privT", article_id="af_priv"), + _article("u", "A.privU"), + ] + + report = _impact(records, articles, "base") + + # The article names the private theorem as its source does, so it is + # impacted and owns the theorem's `where` helper; an exact kernel name + # wins over a private constant with the same user-facing name. + assert _ids(report.statement_impacted) == ["priv", "u"] + assert report.statement_impacted[0].declarations == ("A.privT",) + assert [(helper.name, helper.owner) for helper in report.helpers] == [(f"{privT}.aux", "priv")] + assert report.claim_targets == ("base", "af_priv", "u") + revised = _impact(records, articles, "base", ["A.privT"]) + assert revised.declarations == ("A.privT",) + assert revised.helpers == () + + +def test_a_helper_of_an_unnamed_private_declaration_claims_the_article_naming_its_owner() -> None: + privT = "_private.Demo.B.0.A.privT" + records = _records( + _rec("A.base", "def"), + _rec(privT, module="Demo.B", user_name="A.privT"), + _rec(f"{privT}.aux", module="Demo.B", type_uses=("A.base",), user_name="A.privT.aux", parent=privT), + ) + articles = [_article("base", "A.base"), _article("priv", "A.privT", article_id="af_priv")] + + report = _impact(records, articles, "base") + + assert report.statement_impacted == () + assert [(helper.name, helper.owner) for helper in report.helpers] == [(f"{privT}.aux", "priv")] + assert report.claim_targets == ("base", "af_priv") + + +def test_a_name_several_private_declarations_share_is_refused() -> None: + records = _records( + _rec("A.base", "def"), + _rec("_private.Demo.B.0.A.dup", module="Demo.B", user_name="A.dup"), + _rec("_private.Demo.C.0.A.dup", module="Demo.C", user_name="A.dup"), + ) + articles = [_article("base", "A.base"), _article("dup", "A.dup")] + message = "names several private declarations: _private.Demo.B.0.A.dup, _private.Demo.C.0.A.dup" + + with pytest.raises(ImpactError, match=f"^dup: lean: A\\.dup {message}$"): + _impact(records, articles, "base") + with pytest.raises(ImpactError, match=f"^A\\.dup {message}$"): + _impact(records, articles, "dup") + + +def test_an_alias_shares_its_target_s_statement() -> None: + records = _records( + _rec("A.plain"), + _rec("A.alias", value_uses=("A.plain",), alias_of="A.plain"), + _rec("A.usesAlias", value_uses=("A.alias",)), + _rec("A.statesAlias", type_uses=("A.alias",)), + ) + articles = [ + _article("plain", "A.plain"), + _article("uses-alias", "A.usesAlias"), + _article("states-alias", "A.statesAlias"), + ] + + report = _impact(records, articles, "plain") + + # The alias copied A.plain's type, so revising A.plain revises the alias, + # what states anything about it, and every proof that uses it. + assert _ids(report.statement_impacted) == ["states-alias"] + assert _ids(report.proof_impacted) == ["uses-alias"] + assert [(helper.name, helper.impact) for helper in report.helpers] == [("A.alias", "statement")] + assert report.claim_targets == ("plain", "states-alias", "uses-alias") + + +def test_a_deprecated_constant_s_own_companions_are_not_its_users() -> None: + records = _records( + _rec("A.old", "def", deprecated=True), + _rec("A.old.eq_1", type_uses=("A.old",), uses_deprecated=("A.old",), internal=True, parent="A.old"), + _rec("A.oldThm", deprecated=True), + _rec( + "A.oldThm._simp_1", + value_uses=("A.oldThm",), + uses_deprecated=("A.oldThm",), + internal=True, + parent="A.oldThm", + ), + _rec("A.viaSimp", value_uses=("A.oldThm._simp_1",)), + _rec("A.other"), + ) + articles = [_article("other", "A.other")] + + report = _impact(records, articles, "other") + + # Deleting A.old deletes its equation lemma, but a real user of a + # companion still uses the constant behind it. + assert [(item.name, item.users) for item in report.deprecated] == [("A.old", ()), ("A.oldThm", ("A.viaSimp",))] + assert report.deprecated_unused == ("A.old",) + + +def test_a_deprecated_constant_another_article_names_is_not_unused() -> None: + records = _records( + _rec("A.new"), + _rec("A.old", deprecated=True, replacement="A.new"), + _rec("A.mine", deprecated=True, replacement="A.new"), + _rec("A.used", deprecated=True), + _rec("A.user", value_uses=("A.used",), uses_deprecated=("A.used",)), + ) + articles = [ + _article("new", "A.new", "A.mine"), + _article("b", "A.old"), + _article("a", "A.old", "A.used"), + _article("user", "A.user"), + ] + + report = _impact(records, articles, "new", ["A.new"]) + + # The revised article drops A.mine from lean: in the commit deleting it. + assert [item.as_dict() for item in report.deprecated] == [ + {"name": "A.mine", "replacement": "A.new", "users": [], "articles": []}, + {"name": "A.old", "replacement": "A.new", "users": [], "articles": ["a", "b"]}, + {"name": "A.used", "replacement": None, "users": ["A.user"], "articles": ["a"]}, + ] + assert report.deprecated_unused == ("A.mine",) + assert format_impact(report)[-5:] == [ + "Deprecated:", + " A.mine -> A.new: no users, safe to delete", + " A.old -> A.new: named by a, b", + " A.used: used by A.user; named by a", + "Claim targets: new", + ] + + +def test_helpers_the_revised_article_owns_keep_a_revision_contained() -> None: + records = _records( + _rec("A.S", "inductive", value_uses=("A.S.mk",)), + _rec("A.S.mk", "constructor", type_uses=("A.S",), parent="A.S"), + _rec("A.S.rec", "recursor", type_uses=("A.S", "A.S.mk"), parent="A.S"), + _rec("A.T", "inductive"), + _rec("A.T.aux", type_uses=("A.T",), parent="A.T"), + _rec("A.loose", type_uses=("A.T",)), + ) + articles = [_article("s", "A.S"), _article("t", "A.T"), _article("other", "A.T")] + + owned = _impact(records, articles, "s") + shared = _impact(records, articles, "t") + + # A structure's generated companions are repaired under its own claim and + # are still listed; a helper owned by another article or by none is not. + assert owned.contained + assert [(helper.name, helper.owner) for helper in owned.helpers] == [("A.S.mk", "s"), ("A.S.rec", "s")] + assert owned.claim_targets == ("s",) + assert format_impact(owned)[2] == "Contained: nothing outside s uses A.S, so it can be revised in place." + assert not shared.contained + assert [(helper.name, helper.owner) for helper in shared.helpers] == [("A.T.aux", "other"), ("A.loose", None)] + assert shared.claim_targets == ("t", "other") + + # --------------------------------------------------------------------------- # # Probe output # --------------------------------------------------------------------------- # @@ -340,6 +506,8 @@ def _payload(name: str, **fields: object) -> dict[str, object]: "replacement": None, "uses_deprecated": [], "value_missing": False, + "user_name": None, + "alias_of": None, } record.update(fields) return record @@ -355,6 +523,7 @@ def test_probe_output_is_read_from_marker_lines() -> None: "warning: unrelated Lean output", _line(_payload("A.b", type_uses=["A.a"], parent="A")), _line(_payload("A.a", kind="def", instance=True)), + _line(_payload("_private.Demo.0.A.c", value_uses=["A.b"], user_name="A.c", alias_of="A.b")), "", ] ) @@ -364,7 +533,16 @@ def test_probe_output_is_read_from_marker_lines() -> None: assert records == { "A.a": ConstantRecord(name="A.a", kind="def", module="Demo", instance=True), "A.b": ConstantRecord(name="A.b", kind="theorem", module="Demo", type_uses=("A.a",), parent="A"), + "_private.Demo.0.A.c": ConstantRecord( + name="_private.Demo.0.A.c", + kind="theorem", + module="Demo", + value_uses=("A.b",), + user_name="A.c", + alias_of="A.b", + ), } + assert records["_private.Demo.0.A.c"].meaning_uses == ("A.b",) @pytest.mark.parametrize( @@ -376,6 +554,9 @@ def test_probe_output_is_read_from_marker_lines() -> None: ([_line(_payload("A.a", value_uses=[1]))], "malformed 'value_uses' field"), ([_line(_payload("A.a", kind="lemma"))], "malformed record for 'A.a'"), ([_line(_payload(""))], "malformed record for ''"), + ([_line(_payload("A.a", user_name=""))], "malformed record for 'A.a'"), + ([_line(_payload("A.a", user_name=1))], "malformed 'user_name' field"), + ([_line(_payload("A.a", alias_of="A.b"))], "malformed record for 'A.a'"), ([_line(_payload("A.a")), _line(_payload("A.a"))], "emitted A.a twice"), (["no records here"], "found no project-local constants"), ([_line(_payload("A.a", value_missing=True))], "could not read the value of A.a"), @@ -659,8 +840,8 @@ def test_cli_writes_the_impact_report_as_canonical_json(tmp_path: Path, monkeypa ], "undeclared_dependencies": ["chapter/loose"], "deprecated": [ - {"name": "Demo.gone", "replacement": None, "users": []}, - {"name": "Demo.old", "replacement": "Demo.base_eq", "users": ["Demo.loose"]}, + {"name": "Demo.gone", "replacement": None, "users": [], "articles": []}, + {"name": "Demo.old", "replacement": "Demo.base_eq", "users": ["Demo.loose"], "articles": []}, ], "deprecated_unused": ["Demo.gone"], "claim_targets": [_BASE_ID, _USES_ID, "chapter/loose"], @@ -833,7 +1014,12 @@ def test_helpers_are_located_by_source_name_in_their_module_s_file(tmp_path: Pat monkeypatch, [ *_STUB_RECORDS, - _payload("_private.Demo.Extra.0.Demo.secret", module="Demo.Extra", type_uses=["Demo.base"]), + _payload( + "_private.Demo.Extra.0.Demo.secret", + module="Demo.Extra", + type_uses=["Demo.base"], + user_name="Demo.secret", + ), _payload("Demo.twin", type_uses=["Demo.base"]), _payload("Demo.made", module="Demo.Made", type_uses=["Demo.base"]), _payload("Demo.lost", module="Demo.Made", type_uses=["Demo.base"]), @@ -953,6 +1139,47 @@ def P (n : Nat) : Prop := base n = n """ +_IMP_ALIAS = """\ +import Lean + +open Lean Elab Command + +/-- The theorem branch of Batteries' `alias`: the alias copies the target's +type, and its value is the target constant. -/ +elab "imp_alias " n:ident " := " t:ident : command => do + let target ← liftCoreM <| realizeGlobalConstNoOverloadWithInfo t + let info ← getConstInfo target + liftCoreM <| addDecl <| .thmDecl { info.toConstantVal with + name := (← getCurrNamespace) ++ n.getId + value := mkConst target (info.levelParams.map mkLevelParam) } +""" + +_IMP_MORE = """\ +import Imp.Alias + +namespace Imp + +def seed : Nat := 2 + +private theorem privSeed : seed = 2 := aux +where aux : seed = 2 := rfl + +theorem seedEq : seed = 2 := rfl + +imp_alias seedAlias := seedEq + +theorem usesAlias : seed = 2 ∧ True := ⟨seedAlias, trivial⟩ + +@[simp, deprecated seedEq (since := "2026-10-05")] +def oldSeed : Nat := 2 + +structure Box where + v : Nat + +end Imp +""" + + def _imp_project(root: Path) -> Path: """A built Lean project whose glob selects a module its root never imports.""" @@ -966,6 +1193,8 @@ def _imp_project(root: Path) -> Path: (lean_root / "Imp.lean").write_text("import Imp.Basic\n", encoding="utf-8") (lean_root / "Imp" / "Basic.lean").write_text(_IMP_BASIC, encoding="utf-8") (lean_root / "Imp" / "Extra.lean").write_text(_IMP_EXTRA, encoding="utf-8") + (lean_root / "Imp" / "Alias.lean").write_text(_IMP_ALIAS, encoding="utf-8") + (lean_root / "Imp" / "More.lean").write_text(_IMP_MORE, encoding="utf-8") build = subprocess.run(["lake", "build"], cwd=lean_root, capture_output=True, text=True, timeout=600, check=False) assert build.returncode == 0, build.stdout + build.stderr return lean_root @@ -1025,17 +1254,19 @@ def recorded(*args: object, **kwargs: object) -> str: ("Imp.usesOld", "theorem", "statement", "Imp/Extra.lean", 10, None), ] assert report["undeclared_dependencies"] == ["chapter/simp"] + # The equation lemma `@[simp]` gives oldSeed goes when oldSeed does. assert report["deprecated"] == [ - {"name": "Imp.oldEq", "replacement": "Imp.base_eq", "users": ["Imp.usesOld"]}, - {"name": "Imp.oldUnused", "replacement": "Imp.base_eq", "users": []}, + {"name": "Imp.oldEq", "replacement": "Imp.base_eq", "users": ["Imp.usesOld"], "articles": []}, + {"name": "Imp.oldSeed", "replacement": "Imp.seedEq", "users": [], "articles": []}, + {"name": "Imp.oldUnused", "replacement": "Imp.base_eq", "users": [], "articles": []}, ] - assert report["deprecated_unused"] == ["Imp.oldUnused"] + assert report["deprecated_unused"] == ["Imp.oldSeed", "Imp.oldUnused"] assert report["claim_targets"] == ["chapter/base", "chapter/proved", "chapter/simp", "chapter/uses"] assert report["source_revision"] == source_revision (probed,) = outputs records = parse_impact_output(probed) - assert {record.module for record in records.values()} == {"Imp.Basic", "Imp.Extra"} + assert {record.module for record in records.values()} == {"Imp.Alias", "Imp.Basic", "Imp.Extra", "Imp.More"} simp_lemma = next(record for record in records.values() if record.parent == "Imp.P_iff") assert simp_lemma.internal and simp_lemma.type_uses == ("Imp.P",) assert records["Imp.usesOld"].uses_deprecated == ("Imp.oldEq",) @@ -1056,3 +1287,35 @@ def recorded(*args: object, **kwargs: object) -> str: apart = compute_impact(records, articles, articles[4], ["Imp.apart"], source_revision="rev") assert apart.contained assert apart.claim_targets == ("chapter/apart",) + + # Articles name a private theorem as its source does; its `where` helper + # is owned through the private name. + priv = "_private.Imp.More.0.Imp.privSeed" + assert records[priv].user_name == "Imp.privSeed" + assert records[f"{priv}.aux"].parent == priv + more = [ + ImpactArticle("chapter/seed", None, ("Imp.seed",)), + ImpactArticle("chapter/priv", None, ("Imp.privSeed",)), + ImpactArticle("chapter/seed-eq", None, ("Imp.seedEq",)), + ImpactArticle("chapter/uses-alias", None, ("Imp.usesAlias",)), + ImpactArticle("chapter/box", None, ("Imp.Box",)), + ] + seed = compute_impact(records, more, more[0], ["Imp.seed"], source_revision="rev") + assert _ids(seed.statement_impacted) == ["chapter/priv", "chapter/seed-eq", "chapter/uses-alias"] + assert [(helper.name, helper.owner) for helper in seed.helpers] == [ + ("Imp.seedAlias", None), + (f"{priv}.aux", "chapter/priv"), + ] + assert compute_impact(records, more, more[1], ["Imp.privSeed"], source_revision="rev").contained + # The alias's type copies seedEq's without mentioning it. + assert records["Imp.seedAlias"].alias_of == "Imp.seedEq" + assert "Imp.seedEq" not in records["Imp.seedAlias"].type_uses + seed_eq = compute_impact(records, more, more[2], ["Imp.seedEq"], source_revision="rev") + assert seed_eq.statement_impacted == () + assert _ids(seed_eq.proof_impacted) == ["chapter/uses-alias"] + assert [(helper.name, helper.impact) for helper in seed_eq.helpers] == [("Imp.seedAlias", "statement")] + # A structure's generated companions belong to its own article. + box = compute_impact(records, more, more[4], ["Imp.Box"], source_revision="rev") + assert box.contained + assert box.helpers + assert {helper.owner for helper in box.helpers} == {"chapter/box"} From c75b7bb4c0fefc7f6c5011f4bad5fa050ece803e Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 09:04:55 -0400 Subject: [PATCH 23/38] Refuse imported helper names and quiet lines resting on a failure A root constant named like one of the open probe's helpers made the helper's own definition fail as already declared, so the audit ran with the imported constant in its place. The job still failed on that error, but the log ended with the clean summary. The audit now refuses to start when any helper resolves to an imported constant. A declaration whose `where` clause or other auxiliary failed, or that used a failing helper, still printed a `conditional:` line when it also rested on a declared open statement, because only its own errors suppressed the line. It now gets no clean line when anything it reaches through the audit's edges failed. --- .../templates/github/autoform_audit.py | 54 +++++++++++++++---- .../.github/autoform_audit.py | 54 +++++++++++++++---- tests/test_lake_artifact_audit.py | 47 +++++++++++++++- 3 files changed, 136 insertions(+), 19 deletions(-) diff --git a/autoform_cli/templates/github/autoform_audit.py b/autoform_cli/templates/github/autoform_audit.py index 2d19a210..2e103bcd 100755 --- a/autoform_cli/templates/github/autoform_audit.py +++ b/autoform_cli/templates/github/autoform_audit.py @@ -462,9 +462,10 @@ def render_open_probe( -- No namespace: inside one, Lean resolves a name to a constant under that -- namespace before any other, so a root module could define, say, -- `.Json.parse` and silently replace the parser. At the top level a --- root constant with a helper's name fails as already declared, and one that --- matches a library name makes the reference ambiguous; both fail closed. The --- parser is fully qualified, so a project's own `Json.parse` does not even do that. +-- root constant with a helper's name makes the helper's definition fail as +-- already declared, and the audit below then refuses to run; one that matches a +-- library name makes the reference ambiguous. Both fail closed. The parser is +-- fully qualified, so a project's own `Json.parse` does not even do that. /-- A name spelled as its components: strings, and numbers for numeric ones. -/ def autoformOpenAuditNameOf (json : Json) : Except String Name := do @@ -538,10 +539,39 @@ def autoformOpenAuditEdges (env : Environment) (openSet : Std.HashSet Name) (nod modify (·.insert node hits) return hits +/-- Whether a node rests on a root declaration that failed the audit, through the +same edges as `autoformOpenAuditOpenHits`: a declaration whose `where` clause or +other auxiliary failed gets no line that reads as a clean result. -/ +partial def autoformOpenAuditReachesFailed (env : Environment) (isRoot : Name → Bool) + (openSet : Std.HashSet Name) (failed : Std.HashSet Name) (node : Name) : + StateM (Std.HashMap Name Bool) Bool := do + if let some known := (← get).get? node then + return known + modify (·.insert node false) + let mut broken := failed.contains node + for used in autoformOpenAuditEdges env openSet node do + if broken then + break + if used == ``sorryAx || !isRoot used then + continue + let target := autoformOpenAuditBlockOf env used + if target == node then + continue + broken := failed.contains used || (← autoformOpenAuditReachesFailed env isRoot openSet failed target) + modify (·.insert node broken) + return broken + def autoformOpenAuditNameList (names : Array Name) : MessageData := MessageData.joinSep (names.toList.map MessageData.ofName) ", " run_cmd do + -- A helper whose definition failed as already declared would resolve to the + -- imported constant of that name instead. + for helper in [``autoformOpenAuditNameOf, ``autoformOpenAuditReadArticles, ``autoformOpenAuditBlockOf, + ``autoformOpenAuditEdges, ``autoformOpenAuditOpenHits, ``autoformOpenAuditReachesFailed, + ``autoformOpenAuditNameList] do + if ((← getEnv).getModuleIdxFor? helper).isSome then + throwError "{{helper}} is declared by an imported module instead of this probe; rename that declaration so the audit can run" let targetModules : List Name := [{target_modules}] let allowed : List Name := [``propext, ``Classical.choice, ``Quot.sound] let articles ← match autoformOpenAuditReadArticles {articles} with @@ -565,8 +595,10 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := let mut errors : Array MessageData := #[] let mut hitCache : Std.HashMap Name (Array Name) := {{}} let mut externalSorry : Std.HashMap Name Bool := {{}} - -- Declarations with an error get no info line that reads as a clean result. + -- Declarations with an error, and those resting on one, get no info line that + -- reads as a clean result. let mut failed : Std.HashSet Name := {{}} + let mut brokenCache : Std.HashMap Name Bool := {{}} for declName in roots do let some info := env.find? declName | continue let reported := errors.size @@ -622,9 +654,13 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := else logInfo m!"sorry-free: {{declName}} [{{article}}]" else - let (hits, cache) := Id.run ((autoformOpenAuditOpenHits env isRoot openSet - (autoformOpenAuditBlockOf env declName)).run hitCache) + let block := autoformOpenAuditBlockOf env declName + let (hits, cache) := Id.run ((autoformOpenAuditOpenHits env isRoot openSet block).run hitCache) hitCache := cache + let (reachesFailed, cache) := Id.run + ((autoformOpenAuditReachesFailed env isRoot openSet failed block).run brokenCache) + brokenCache := cache + let broken := failed.contains declName || reachesFailed let undeclared := hits.filter (fun hit => !allowedOpen.contains hit) unless undeclared.isEmpty do errors := errors.push m!"{{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList undeclared}}, which its article's Markdown dependencies do not reach; add the dependency to the article or stop using them" @@ -639,14 +675,14 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := logInfo m!"open statement (proof is sorry): {{declName}} [{{article}}]" else if (← Lean.collectAxioms declName).contains ``sorryAx then logInfo m!"open statement (proof depends on sorry elsewhere): {{declName}} [{{article}}]" - else if errors.size == reported && !failed.contains declName then + else if errors.size == reported && !broken then logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" else if !hits.isEmpty then conditionalCount := conditionalCount + 1 - if errors.size == reported && !failed.contains declName then + if errors.size == reported && !broken then logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList hits}}" else unless (← Lean.collectAxioms declName).contains ``sorryAx do - if errors.size == reported && !failed.contains declName then + if errors.size == reported && !broken then logInfo m!"sorry-free: {{declName}} [{{article}}]" for error in errors do logError error diff --git a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py index 2d19a210..2e103bcd 100755 --- a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py +++ b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py @@ -462,9 +462,10 @@ def render_open_probe( -- No namespace: inside one, Lean resolves a name to a constant under that -- namespace before any other, so a root module could define, say, -- `.Json.parse` and silently replace the parser. At the top level a --- root constant with a helper's name fails as already declared, and one that --- matches a library name makes the reference ambiguous; both fail closed. The --- parser is fully qualified, so a project's own `Json.parse` does not even do that. +-- root constant with a helper's name makes the helper's definition fail as +-- already declared, and the audit below then refuses to run; one that matches a +-- library name makes the reference ambiguous. Both fail closed. The parser is +-- fully qualified, so a project's own `Json.parse` does not even do that. /-- A name spelled as its components: strings, and numbers for numeric ones. -/ def autoformOpenAuditNameOf (json : Json) : Except String Name := do @@ -538,10 +539,39 @@ def autoformOpenAuditEdges (env : Environment) (openSet : Std.HashSet Name) (nod modify (·.insert node hits) return hits +/-- Whether a node rests on a root declaration that failed the audit, through the +same edges as `autoformOpenAuditOpenHits`: a declaration whose `where` clause or +other auxiliary failed gets no line that reads as a clean result. -/ +partial def autoformOpenAuditReachesFailed (env : Environment) (isRoot : Name → Bool) + (openSet : Std.HashSet Name) (failed : Std.HashSet Name) (node : Name) : + StateM (Std.HashMap Name Bool) Bool := do + if let some known := (← get).get? node then + return known + modify (·.insert node false) + let mut broken := failed.contains node + for used in autoformOpenAuditEdges env openSet node do + if broken then + break + if used == ``sorryAx || !isRoot used then + continue + let target := autoformOpenAuditBlockOf env used + if target == node then + continue + broken := failed.contains used || (← autoformOpenAuditReachesFailed env isRoot openSet failed target) + modify (·.insert node broken) + return broken + def autoformOpenAuditNameList (names : Array Name) : MessageData := MessageData.joinSep (names.toList.map MessageData.ofName) ", " run_cmd do + -- A helper whose definition failed as already declared would resolve to the + -- imported constant of that name instead. + for helper in [``autoformOpenAuditNameOf, ``autoformOpenAuditReadArticles, ``autoformOpenAuditBlockOf, + ``autoformOpenAuditEdges, ``autoformOpenAuditOpenHits, ``autoformOpenAuditReachesFailed, + ``autoformOpenAuditNameList] do + if ((← getEnv).getModuleIdxFor? helper).isSome then + throwError "{{helper}} is declared by an imported module instead of this probe; rename that declaration so the audit can run" let targetModules : List Name := [{target_modules}] let allowed : List Name := [``propext, ``Classical.choice, ``Quot.sound] let articles ← match autoformOpenAuditReadArticles {articles} with @@ -565,8 +595,10 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := let mut errors : Array MessageData := #[] let mut hitCache : Std.HashMap Name (Array Name) := {{}} let mut externalSorry : Std.HashMap Name Bool := {{}} - -- Declarations with an error get no info line that reads as a clean result. + -- Declarations with an error, and those resting on one, get no info line that + -- reads as a clean result. let mut failed : Std.HashSet Name := {{}} + let mut brokenCache : Std.HashMap Name Bool := {{}} for declName in roots do let some info := env.find? declName | continue let reported := errors.size @@ -622,9 +654,13 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := else logInfo m!"sorry-free: {{declName}} [{{article}}]" else - let (hits, cache) := Id.run ((autoformOpenAuditOpenHits env isRoot openSet - (autoformOpenAuditBlockOf env declName)).run hitCache) + let block := autoformOpenAuditBlockOf env declName + let (hits, cache) := Id.run ((autoformOpenAuditOpenHits env isRoot openSet block).run hitCache) hitCache := cache + let (reachesFailed, cache) := Id.run + ((autoformOpenAuditReachesFailed env isRoot openSet failed block).run brokenCache) + brokenCache := cache + let broken := failed.contains declName || reachesFailed let undeclared := hits.filter (fun hit => !allowedOpen.contains hit) unless undeclared.isEmpty do errors := errors.push m!"{{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList undeclared}}, which its article's Markdown dependencies do not reach; add the dependency to the article or stop using them" @@ -639,14 +675,14 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := logInfo m!"open statement (proof is sorry): {{declName}} [{{article}}]" else if (← Lean.collectAxioms declName).contains ``sorryAx then logInfo m!"open statement (proof depends on sorry elsewhere): {{declName}} [{{article}}]" - else if errors.size == reported && !failed.contains declName then + else if errors.size == reported && !broken then logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" else if !hits.isEmpty then conditionalCount := conditionalCount + 1 - if errors.size == reported && !failed.contains declName then + if errors.size == reported && !broken then logInfo m!"conditional: {{declName}} [{{article}}] rests on open statement(s) {{autoformOpenAuditNameList hits}}" else unless (← Lean.collectAxioms declName).contains ``sorryAx do - if errors.size == reported && !failed.contains declName then + if errors.size == reported && !broken then logInfo m!"sorry-free: {{declName}} [{{article}}]" for error in errors do logError error diff --git a/tests/test_lake_artifact_audit.py b/tests/test_lake_artifact_audit.py index 91a487a7..6a1e25d4 100644 --- a/tests/test_lake_artifact_audit.py +++ b/tests/test_lake_artifact_audit.py @@ -697,6 +697,11 @@ def run(*arguments: str) -> subprocess.CompletedProcess[str]: theorem native_open : 3 + 3 = 6 := by native_decide +theorem where_conditional : 1 + 1 = 2 ∧ True := ⟨open_stmt, aux⟩ +where aux : True := sorry + +theorem helper_conditional : 1 + 1 = 2 ∧ True := ⟨open_stmt, helper_sorry⟩ + end Fixture """ @@ -717,6 +722,22 @@ def {parser} (_ : String) : Except String Lean.Json := Lean.Json.parse {table} """ +# A root constant with a helper's name, so the probe's own definition fails. +_CLASH_LEAN = """import Lean.Data.Json + +namespace Fixture + +theorem cheat : 2 + 2 = 5 := sorry + +theorem fully : 2 + 2 = 5 := cheat + +end Fixture + +def autoformOpenAuditReadArticles (_ : String) : + Except String (Array (Lean.Name × String × Bool × Array Lean.Name)) := + .ok #[(`Fixture.cheat, "x", true, #[`Fixture.cheat])] +""" + @pytest.fixture(scope="module") def open_projects(tmp_path_factory: pytest.TempPathFactory) -> dict[str, tuple[Path, Path]]: @@ -735,6 +756,7 @@ def open_projects(tmp_path_factory: pytest.TempPathFactory) -> dict[str, tuple[P # precedence over the library parser. ("hijack", _FORGED_LEAN.format(parser="AutoformOpenStatementAudit.Json.parse", table=_FORGED_TABLE)), ("forged", _FORGED_LEAN.format(parser="Json.parse", table=_FORGED_TABLE)), + ("clash", _CLASH_LEAN), ): project = root / name _write(project / "lean-toolchain", "leanprover/lean4:v4.32.2\n") @@ -869,6 +891,8 @@ def test_open_probe_logs_no_clean_result_for_a_failing_declaration( _article("uses-native", ["Fixture.uses_native"]), _article("native-reduction", ["Fixture.native_reduction"], allowed=["Fixture.open_stmt"]), _article("native-open", ["Fixture.native_open"], is_open=True, allowed=["Fixture.native_open"]), + _article("where-conditional", ["Fixture.where_conditional"], allowed=["Fixture.open_stmt"]), + _article("helper-conditional", ["Fixture.helper_conditional"], allowed=["Fixture.open_stmt"]), _article("clean", ["Fixture.clean"]), ), ) @@ -880,11 +904,14 @@ def test_open_probe_logs_no_clean_result_for_a_failing_declaration( "Fixture.uses_native depends on unexpected axiom", "Fixture.native_reduction depends on unexpected axiom", "Fixture.native_open depends on unexpected axiom", + "Fixture.where_conditional.aux contains sorry but is not an open statement", + "Fixture.helper_sorry contains sorry but is not an open statement", "open statement (proof is sorry): Fixture.open_stmt [open]", "sorry-free: Fixture.clean [clean]", ): assert message in output - for name in ("reduction", "uses_native", "native_reduction", "native_open"): + names = ("reduction", "uses_native", "native_reduction", "native_open", "where_conditional", "helper_conditional") + for name in names: for line in output.splitlines(): if f"Fixture.{name} [" in line: assert not line.startswith(("sorry-free:", "conditional:", "open statement (")), line @@ -907,6 +934,24 @@ def test_open_probe_reads_its_table_with_the_library_parser( assert "kernel trust clean" not in output +def test_open_probe_refuses_a_helper_name_an_imported_module_declares( + helper: ModuleType, open_projects: dict[str, tuple[Path, Path]] +) -> None: + audited = _audit( + helper, + open_projects["clash"], + _contract(_article("fully", ["Fixture.fully"]), _article("cheat", ["Fixture.cheat"])), + ) + + output = audited.stdout + audited.stderr + assert audited.returncode != 0, output + assert ( + "autoformOpenAuditReadArticles is declared by an imported module instead of this probe; " + "rename that declaration so the audit can run" + ) in output + assert "kernel trust clean" not in output + + def test_strict_probe_still_rejects_every_sorry( helper: ModuleType, open_projects: dict[str, tuple[Path, Path]] ) -> None: From 59d189e2c445cd445ccb4d2cae4ea1ad11cf93ff Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 09:12:03 -0400 Subject: [PATCH 24/38] Count a theorem as an alias only when it copies its target's type work impact treated any theorem whose value is exactly another local constant as that constant's alias and put it in the statement-impacted set with its target. A hand-written theorem such as `theorem t : seed = 1 + 1 := seedEq` keeps its own statement when seedEq's changes; only its proof breaks. Require the theorem's type to be exactly the target's, instantiated at the value's universe levels, which is what Batteries' alias writes, so such a theorem is reported as proof-impacted instead. --- autoform_cli/impact.py | 10 +++++----- autoform_cli/probes/impact_probe.lean | 15 ++++++++++----- tests/test_impact.py | 11 ++++++++++- 3 files changed, 25 insertions(+), 11 deletions(-) diff --git a/autoform_cli/impact.py b/autoform_cli/impact.py index aefbc355..08a7abaa 100644 --- a/autoform_cli/impact.py +++ b/autoform_cli/impact.py @@ -56,8 +56,8 @@ class ConstantRecord: Lean generates (``_proof_1``, ``match_1``, ``_simp_1``) are. ``user_name`` is the user-facing name of a private constant, which is how articles and ``--declaration`` name it, and ``None`` for any other. - ``alias_of`` is the local constant a theorem's value is exactly, as for - Batteries' ``alias``, whose type is copied from that constant's. + ``alias_of`` is the local constant a theorem's value is exactly when its + type is exactly that constant's too, as Batteries' ``alias`` writes. """ name: str @@ -447,9 +447,9 @@ def compute_impact( directly or through internal-detail constants, such as the ``_simp_1`` companion ``simp`` uses or the ``_proof_1`` a definition's nested proof becomes. Definitions belong in P too: their nested proofs can stop - elaborating although their meaning is unchanged. A theorem whose value is - exactly another constant, as Batteries' ``alias`` makes, has that - constant's statement, so it joins M with it. + elaborating although their meaning is unchanged. A theorem whose value and + type are exactly another constant and its type, as Batteries' ``alias`` + writes, has that constant's statement, so it joins M with it. """ resolve = _resolver(records) diff --git a/autoform_cli/probes/impact_probe.lean b/autoform_cli/probes/impact_probe.lean index a43e59c2..8bf28cf2 100644 --- a/autoform_cli/probes/impact_probe.lean +++ b/autoform_cli/probes/impact_probe.lean @@ -59,9 +59,11 @@ statements they use change; an inductive's constructors stand in for a value. wrote is not an internal detail, while the companions Lean generates for it (`_proof_1`, `match_1`, `_simp_1`) still are. `user_name` is that name for a private constant, which is how articles and `--declaration` name it. -`alias_of` is the local constant a theorem's value is exactly, as for -Batteries' `alias`: such a theorem copies its target's type, so its statement -changes with the target's although its type never mentions it. -/ +`alias_of` is the local constant a theorem's value is exactly when the +theorem's type is also exactly that constant's, as Batteries' `alias` writes: +its statement then changes with the target's although its type never mentions +it. A theorem whose type differs from its target's, even up to definitional +unfolding, keeps its own statement. -/ def record (env : Environment) (isLocal : Name → Bool) (c : Name) (info : ConstantInfo) (module : Name) : Json := let value := info.value? (allowOpaque := true) @@ -73,8 +75,11 @@ def record (env : Environment) (isLocal : Name → Bool) (c : Name) (info : Cons | .thmInfo _ | .defnInfo _ | .opaqueInfo _ => value.isNone | _ => false let aliasOf := match info, value.map (·.consumeMData) with - | .thmInfo _, some (.const target _) => if isLocal target then some target else none - | _, _ => none + | .thmInfo _, some (.const target levels) => + let sameType := (env.find? target).any + (fun (targetInfo : ConstantInfo) => info.type == targetInfo.instantiateTypeLevelParams levels) + if isLocal target && sameType then some target else none + | _, _ => none let usesDeprecated := (typeConstants ++ valueConstants).foldl (fun acc d => if Linter.isDeprecated env d && !acc.contains d then acc.push d else acc) #[] Json.mkObj [ diff --git a/tests/test_impact.py b/tests/test_impact.py index 82d6c106..c7b00d87 100644 --- a/tests/test_impact.py +++ b/tests/test_impact.py @@ -1168,6 +1168,8 @@ def seed : Nat := 2 imp_alias seedAlias := seedEq +theorem seedAgain : seed = 1 + 1 := seedEq + theorem usesAlias : seed = 2 ∧ True := ⟨seedAlias, trivial⟩ @[simp, deprecated seedEq (since := "2026-10-05")] @@ -1303,6 +1305,7 @@ def recorded(*args: object, **kwargs: object) -> str: seed = compute_impact(records, more, more[0], ["Imp.seed"], source_revision="rev") assert _ids(seed.statement_impacted) == ["chapter/priv", "chapter/seed-eq", "chapter/uses-alias"] assert [(helper.name, helper.owner) for helper in seed.helpers] == [ + ("Imp.seedAgain", None), ("Imp.seedAlias", None), (f"{priv}.aux", "chapter/priv"), ] @@ -1310,10 +1313,16 @@ def recorded(*args: object, **kwargs: object) -> str: # The alias's type copies seedEq's without mentioning it. assert records["Imp.seedAlias"].alias_of == "Imp.seedEq" assert "Imp.seedEq" not in records["Imp.seedAlias"].type_uses + # A theorem whose type only unfolds to its target's states its own type. + assert records["Imp.seedAgain"].alias_of is None + assert records["Imp.seedAgain"].value_uses == ("Imp.seedEq",) seed_eq = compute_impact(records, more, more[2], ["Imp.seedEq"], source_revision="rev") assert seed_eq.statement_impacted == () assert _ids(seed_eq.proof_impacted) == ["chapter/uses-alias"] - assert [(helper.name, helper.impact) for helper in seed_eq.helpers] == [("Imp.seedAlias", "statement")] + assert [(helper.name, helper.impact) for helper in seed_eq.helpers] == [ + ("Imp.seedAgain", "proof"), + ("Imp.seedAlias", "statement"), + ] # A structure's generated companions belong to its own article. box = compute_impact(records, more, more[4], ["Imp.Box"], source_revision="rev") assert box.contained From 34202082bdd2a0957b9d56411afa31667b1c8c2f Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 09:13:50 -0400 Subject: [PATCH 25/38] Document open statements under revision and the impact report's limits Bring the README and the Formalize, Roadmap and Human Review skills in line with the code: a definition is ready to state only once its proof prerequisites are stated, a retracted theorem stays open while its lean: names the old declaration, the open audit logs at most one line per declaration and none that reads as passing after an error, and a recursive proof must be sorry as a whole. Describe how work impact names private declarations, what counts as an alias, and which derived declarations it does not report. --- autoform_cli/README.md | 147 +++++++++++++++++++++++------------ skills/formalize/SKILL.md | 41 ++++++---- skills/human-review/SKILL.md | 10 +-- skills/roadmap/SKILL.md | 20 +++-- 4 files changed, 139 insertions(+), 79 deletions(-) diff --git a/autoform_cli/README.md b/autoform_cli/README.md index dabae359..468b052e 100644 --- a/autoform_cli/README.md +++ b/autoform_cli/README.md @@ -100,7 +100,7 @@ lands only with its proof; under `open_statements: allowed` it may land with a | Derived state | Holds when | | --- | --- | -| `can_state` | Every statement prerequisite is stated and, under the strict policy, every proof prerequisite is proved. | +| `can_state` | Every statement prerequisite is stated and every proof prerequisite is proved. Under the open policy a theorem waits for no proof prerequisite, and a definition, whose body is its proof, waits for them to be stated. | | `can_prove` | Stated, every statement prerequisite is stated, and every proof prerequisite is proved (strict policy) or stated (open policy). | | `proved` | The proof compiles. | | `conditional` | Proved, but the proof rests on an open statement; open policy only. | @@ -112,7 +112,8 @@ compiles but which rests on unfinished work is green, not dark green. `conditional`, labelled "conditionally proved", is the open policy's case of that: the article records `proof: formalized` and its proof compiles, but it reaches an open statement, an article whose statement is formalized and whose -proof is not, through its dependencies. An open dependency counts together with +proof is not, or a retracted theorem (see [Open statements](#open-statements)), +through its dependencies. An open dependency counts together with whatever its statement prerequisites reach, and a proved dependency passes on everything it reaches. The site colours it violet, never green, and lists those open statements in an `Assumes` row on the article page. It is never @@ -514,9 +515,10 @@ is unblocked, by the same policy-dependent rule as the derived `can_state` and `can_prove` states, the runtime projection, and the site's Next up card. Under the strict policy project CI rejects `sorry`, so a theorem's statement lands with its proof, and an article's statement phase also waits until its `## Proof -depends on` prerequisites are proved. Under `open_statements: allowed` the -statement phase waits only for the statement prerequisites to be stated, and -the proof phase for every prerequisite to be stated. `work +depends on` prerequisites are proved. Under `open_statements: allowed` a +theorem's statement phase waits only for the statement prerequisites to be +stated, a definition's also for its proof prerequisites, which its body uses, +and the proof phase for every prerequisite to be stated. `work context` accepts the path-derived node ID (see Articles and containment) or an assigned `article_id` and reports the exact article, dependencies, source targets, Lean targets, blockers, article and graph source revisions, and claim @@ -551,13 +553,14 @@ autoform work assumptions blueprint --json ``` `work assumptions` prints the policy, one `open:` line per open statement with -its declarations, and one `conditional:` line per article whose Lean rests on -open statements. `--json` writes the `autoform-assumptions/v1` contract that CI -audits the build against: every stated article whose `lean:` names a -declaration, with `open`, `assumes`, and `allowed_open_declarations`, the -declarations of the open statements its Lean may reach, plus its own when it is -open. Under the strict policy every such article is listed with `open` false -and nothing allowed. It reads Markdown only and needs no Lean build. +its declarations and the open statements it assumes, if any, and one +`conditional:` line per other article whose Lean rests on open statements. +`--json` writes the `autoform-assumptions/v1` contract that CI audits the build +against: every article whose `lean:` names a declaration, stated or not, except +`mathlib: true` ones, with `open`, `assumes`, and `allowed_open_declarations`, +the declarations of the open statements its Lean may reach, plus its own when +it is open. Under the strict policy every such article is listed with `open` +false and nothing allowed. It reads Markdown only and needs no Lean build. Ask what revising an article's Lean declarations would affect before editing them: @@ -568,29 +571,42 @@ autoform work impact chapter/result . --lean-root . --declaration MyProject.help ``` The revised set is the article's `lean:` declarations, or the `--declaration` -names, each of which must be a project-local constant. `work impact` runs a -Lean probe against the built project through the same freshness check, output -bound, and process handling as `skeleton`, so it needs a fresh `lake build` -and a writable `.lake`, and it carries the same trust caveat: run it only in a -trusted checkout or a sandbox. `--timeout SECONDS` sets the probe's budget, -600 seconds by default. The report lists: +names, each of which must be a project-local constant; a private declaration +goes by the name its source gives it, and `work impact` refuses to run while an +article's `lean:` or a `--declaration` names several private declarations at +once. `work impact` runs a Lean probe against the built project through the +same freshness check, output bound, and process handling as `skeleton`, so it +needs a fresh `lake build` and a writable `.lake`, and it carries the same +trust caveat: run it only in a trusted checkout or a sandbox. `--timeout +SECONDS` sets the probe's budget, 600 seconds by default. The report lists: - statement-impacted articles, whose declarations' meaning (a type, or a - definition's body or an inductive's constructors) reaches the revised set; + definition's body or an inductive's constructors) reaches the revised set, + where a theorem whose type is exactly another constant's and whose proof is + that constant, as `alias` writes, shares that constant's meaning, even when + written by hand; - proof-impacted articles, whose proofs or definition bodies use the revised set, directly or through Lean-generated companions such as the `_simp_1` lemma `simp` uses, without their meaning changing; - helpers that no article names, each with an owner when one exists: the article naming its nearest ancestor by name, such as `Foo` for `Foo.aux`; - impacted articles without a Markdown dependency path to the revised article; -- every deprecated project declaration with its replacement and users, and in - `deprecated_unused` those nothing uses; +- every deprecated project declaration with its replacement, its users, not + counting companions Lean generates for it such as `X.eq_1`, and the other + articles whose `lean:` names it, and in `deprecated_unused` those with no + users and no such article; - the claim targets: the revised article's first, then every impacted - article's. - -A revision nothing else uses is `contained` and can be made in place. `--json` -writes `autoform-impact/v1`. The [revision contract](#revision-contract) says -what to do with the answer. + article's and every helper owner's. + +A revision is `contained` when no other article uses it and every helper it +impacts belongs to the revised article; it can then be made in place under +that article's claim. A declaration derived from a revised one without naming +it in its statement changes with it, but `work impact` does not report that +declaration's users: the additive form `to_additive` writes, which goes +unreported itself, or a direction `alias ⟨mp, mpr⟩ :=` takes of an `Iff`, which +shows only as proof-impacted. Check such derivations on the revised set by +hand. `--json` writes `autoform-impact/v1`. The [revision +contract](#revision-contract) says what to do with the answer. Plan durable article identity metadata without changing the blueprint: @@ -730,15 +746,28 @@ proved, never as fully proved. The policy lets dependents be stated and proved against a faithful statement before its proof exists; the price is conditional results that stay incomplete until every open statement they rest on is proved. +A theorem that loses `statement` while its `lean:` still names a declaration, +as a retraction or a revision leaves it, stays an open statement: that Lean, +`sorry` or not, still compiles into whatever uses it. The audit keeps accepting +its `sorry`, and what rests on it stays conditional, until its statement is +recorded again or its `lean:` is removed. A definition is never open: its body +is its proof, CI rejects a `sorry` in it, and its statement phase waits until +its proof prerequisites are stated. Turn the policy back off only once no open +statement remains, since the strict audit rejects every `sorry` and strict +status shows a proof resting on one as proved, not conditional. + Write an open statement's proof as exactly `sorry`. The audit accepts a `sorry` only inside the proof of a theorem that an open article's `lean:` names: never in its type, a helper, a definition, or a `where` clause, and never inherited from a declaration outside the root package. Lean-generated auxiliaries count as -helpers: a `where` clause becomes `T.aux`, and well-founded recursion over two -or more arguments moves the `decreasing_by` proof into `T._unary`, so both -fail. `lake build --wfail` and `warningAsError` turn Lean's "declaration uses -`sorry`" warning into an error, so they cannot be combined with open -statements; the generated workflow runs plain `lake build`. +helpers: a `where` clause becomes `T.aux`, well-founded recursion over two or +more arguments moves the `decreasing_by` proof into `T._unary`, and structural +recursion through a `mutual` block compiles the bodies into `T._f`, so all three +fail. Recursion can move a `sorry` case into such an auxiliary, so write the +whole proof as `sorry`, never one case of it. `lake build --wfail` and +`warningAsError` turn Lean's "declaration uses `sorry`" warning into an error, +so they cannot be combined with open statements; the generated workflow runs +plain `lake build`. The generated `autoform-verify.yml` reads the policy with `python3 .github/autoform_audit.py --policy blueprint`, which prints `allowed` or @@ -756,8 +785,9 @@ that depends on `sorry`, an article declaration that reaches an open statement its Markdown dependencies do not reach (so a fully proved article, which assumes nothing, may reach none), a `lean:` name missing from the build, and an open statement that its article records as proved. Each article -declaration gets one log line, with `NAME` the declaration and `ID` the -article's node ID: +declaration gets at most one log line, with `NAME` the declaration and `ID` the +article's node ID, and one with an error gets none of the last three, which +read as passing: ```text open statement (proof is sorry): NAME [ID] @@ -834,7 +864,8 @@ failure behavior must be removed before they use the canonical claim API. Revising a declaration X of article R that other articles' Lean uses touches work R's claim does not cover. This contract makes that work claimable and -keeps the default build passing. +keeps the default build passing. Roadmap records a requested revision in the +Markdown (step 6); Formalize carries out the Lean side (steps 1 to 5). 1. On a fresh build, run `autoform work impact R . --lean-root .`, with `--declaration` when only some of R's declarations, or a helper, change. @@ -843,34 +874,52 @@ keeps the default build passing. - **Expand, migrate, contract**, the default whenever anything uses X: add X' with the revised statement, leave X unchanged and mark it `@[deprecated X' (since := "YYYY-MM-DD")]`, and point R's `lean:` at X'. + Under the open policy, while X's proof is still `sorry`, R's `lean:` + names X beside X', so the audit keeps accepting that `sorry` as an open + statement, and R records `proof` only after step 4 deletes X. Statement-impacted articles lose `statement` and `proof` but keep `lean:`, - so they return to the frontier. Proof-impacted articles keep everything, + so they return to the frontier; under the open policy a statement-impacted + theorem stays an open statement meanwhile (see [open + statements](#open-statements)). Proof-impacted articles keep everything, since their proofs still use the valid old X; migrating them to X' is later work. Claim R and the statement-impacted articles, whose frontmatter changes. - **In place**, only when X and X' cannot coexist, for example an instance - or a structure change: claim every `claim_targets` entry plus the claim - target of every helper's owner, and repair every impacted declaration in - one commit whose default build passes. A repaired dependent proof keeps - `proof: formalized` only after an Agent Review of the repair; otherwise - retract it. Under the open policy, replace its proof with `sorry` and - remove `proof`; under the strict policy, remove the declaration and its - assertions only if nothing else uses it, and use expand, migrate, - contract otherwise. Record what happened under `## Execution notes` of - each touched article. + or a structure change: claim every `claim_targets` entry and repair every + impacted declaration in one commit whose default build passes. A + statement-impacted article keeps `statement` only after an Agent Review of + its source faithfulness under X's new meaning; otherwise it loses + `statement` and `proof` but keeps `lean:`. A repaired dependent proof + keeps `proof: formalized` only after an Agent Review of the repair; + otherwise it loses `proof`. A theorem's proof that cannot be repaired + becomes exactly `sorry` under the open policy; otherwise delete the + declaration and remove its article's `lean:`, `statement`, and `proof`, + which works only when nothing else uses it. When neither applies, the + revision is blocked: release the claims and report it. Record what + happened under `## Execution notes` of each touched article. 3. Claim the whole set with one `autoform claim acquire`. When it is refused, release everything and report the held claim as the blocker. After acquiring, re-run `work impact`; if the set grew, release and start over - with the larger set. Land one commit, then release every claim. -4. Contract: delete a deprecated X once `work impact R . --lean-root . - --declaration X` reports it contained, or X appears in `deprecated_unused`. + with the larger set. Under the open policy, reproduce the CI audit as [open + statements](#open-statements) shows before landing. Land one commit, then + release every claim. +4. Contract: delete a deprecated X once it appears in `deprecated_unused` of + `work impact R . --lean-root .`, meaning no declaration uses it and no + other article's `lean:` names it, and drop it from R's `lean:` in the same + commit. `contained` is not enough: it ignores R's own declarations, such as + an X' built from X. 5. `autoform audit --lean-root` reports `lean-target-deprecated` for an article - whose `lean:` names a declaration carrying the `deprecated` attribute; point - it at the replacement. + whose `lean:` names a declaration with `deprecated` in its own `@[...]` + attribute list; point it at the replacement. The check is lexical, so it + misses a later `attribute [deprecated] X`, which the deprecated list of + `work impact` does see. Under the open policy the finding is expected for a + superseded X that step 2 keeps in R's `lean:` until step 4. 6. When Roadmap revises an article's statement text, it removes that article's `statement` and `proof` but keeps `lean:`, which `work impact` needs. It retracts only that article and the dependents whose Markdown text the revision rewrites; the Lean-side impact decides every other dependent. + Roadmap edits only Markdown: it records the decision, releases its claims, + and leaves the Lean revision to Formalize. ## Local runtime doctor diff --git a/skills/formalize/SKILL.md b/skills/formalize/SKILL.md index 0998f4d3..3e15fbe4 100644 --- a/skills/formalize/SKILL.md +++ b/skills/formalize/SKILL.md @@ -63,7 +63,9 @@ phase: do not modify another article or its Lean declarations, or weaken a public statement. The one exception is revising a declaration other articles' Lean uses: start from `autoform work impact` and make only the edits the [revision contract](../../autoform_cli/README.md#revision-contract) requires, -under the claims it requires. Search the pinned Mathlib checkout before adding +under the claims it requires. A statement phase whose `lean:` already names a +compiled declaration that other articles use, as a Roadmap retraction leaves +it, is such a revision. Search the pinned Mathlib checkout before adding helpers, and use the shared Lean LSP and REPL with `` as the project path. Finish with the focused Lake target. Declare the result in a module the library root imports, or that the lakefile's globs cover, because the default @@ -76,24 +78,29 @@ open-statement policy below allows. project CI rejects `sorry`: the statement phase writes the declaration and, for a theorem, the complete proof, which is why `work list` offers the phase only once the proof prerequisites are proved. Under `open_statements: allowed`, the -statement phase writes the faithful statement with a proof body of exactly -`sorry`, or the full proof, and records only `statement: formalized`. That -`sorry` goes in the declaration's own body, never in its type, a helper, a -definition, or a `where` clause. In both policies the proof phase completes the +statement phase of a theorem writes the faithful statement with a proof body of +exactly `sorry` and records `statement: formalized`, or writes the full proof +and, on acceptance, records both assertions. That `sorry` is the declaration's +whole body: never part of its type, a helper, a definition, or a `where` clause, +and never one case of a recursive proof, which Lean can compile into +auxiliaries such as `_f` that CI rejects. A definition is never left open: its +body is its proof, so `work list` offers its statement phase only once its +proof prerequisites are stated. In both policies the proof phase completes the proof of an already recorded statement without changing that statement. -Under the open policy a proof may use the open statements of its declared -dependencies, direct or transitive. The article then shows as conditionally -proved, and `#print axioms` lists the `sorryAx` it inherits from them without -saying from where. `autoform work assumptions` lists the open statements the -Markdown lets this article assume. Before landing, record `proof: formalized` -in the worktree (until then the audit treats the article as open), reproduce -the CI audit as the [open statements -reference](../../autoform_cli/README.md#open-statements) shows, and require a -`conditional` or `sorry-free` line for each recorded declaration and a passing -summary. An open statement the Markdown does not declare as a dependency fails -CI: send the missing dependency to Roadmap or stop using it. Never describe a -conditional proof as complete, fully proved, or sorry-free. +Under the open policy a proof may use the open statements its Markdown +dependencies reach: each open dependency with whatever its statement +prerequisites reach, and everything a proved dependency reaches, but not an +open dependency's proof prerequisites. `autoform work assumptions --json` lists +the exact `allowed_open_declarations`. The article then shows as conditionally +proved, and `#print axioms` lists the `sorryAx` it inherits without saying from +where. Before landing, record `proof: formalized` in the worktree (until then +the audit treats the article as open), reproduce the CI audit as the [open +statements reference](../../autoform_cli/README.md#open-statements) shows, and +require a `conditional` or `sorry-free` line for each recorded declaration and +a passing summary. An open statement the Markdown does not declare as a +dependency fails CI: send the missing dependency to Roadmap or stop using it. +Never describe a conditional proof as complete, fully proved, or sorry-free. After the focused build passes, require an independent Agent Review of every changed statement or proof for source faithfulness, dependency correctness, and diff --git a/skills/human-review/SKILL.md b/skills/human-review/SKILL.md index 553c6e15..7dd118a7 100644 --- a/skills/human-review/SKILL.md +++ b/skills/human-review/SKILL.md @@ -37,9 +37,9 @@ individual node and Lean-source links. Record each human decision as `approve`, `block`, with the exact page or node and rationale. Separate validator output from the person's judgment. Do not silently apply requested revisions: hand mathematical-plan changes and Lean implementation changes to Roadmap, which -records the decision and follows the [revision -contract](../../autoform_cli/README.md#revision-contract) for declarations -other articles use, and autonomous rubric scoring to Agent Review. +records the decision and leaves the Lean side to Formalize under the [revision +contract](../../autoform_cli/README.md#revision-contract), and autonomous +rubric scoring to Agent Review. Treat the landing page's `Scoped roadmap` percentage as completion among formalizable leaf targets that are fully proved, including every dependency @@ -50,8 +50,8 @@ Mathlib. Treat the percentage never as whole-source completion. Read the adjacent declared source coverage and its linked coverage contract before making scope claims. A statement-only theorem remains incomplete whether it is blocked or ready to prove. In a project that allows open statements, a -conditionally proved target, whose proof rests on an open statement whose Lean -proof is still `sorry`, is not complete either: it is never fully proved, so it +conditionally proved target, whose proof rests on an open statement without a +recorded Lean proof, is not complete either: it is never fully proved, so it counts toward the percentage's total but not its completed share, as does every target that depends on it. Present it as conditional, naming the open statements in its `Assumes` row, never as proved or sorry-free. diff --git a/skills/roadmap/SKILL.md b/skills/roadmap/SKILL.md index 10e3735e..59b0d33d 100644 --- a/skills/roadmap/SKILL.md +++ b/skills/roadmap/SKILL.md @@ -76,14 +76,16 @@ from. A refused acquire means another agent owns the article: leave it and report it. Claims write refs to the board's remote, which is outward-facing, so make sure the request covers them. When a revision changes a statement, remove its `statement` and `proof` metadata but keep `lean:`, which Formalize needs to -run `autoform work impact`. Retract only that article and the dependents whose -Markdown text the revision rewrites, claiming them all in one acquire; the -Lean-side impact decides every other dependent. For a Lean revision requested -in Human Review, record the decision in the article, then follow the [revision -contract](../../autoform_cli/README.md#revision-contract): run `work impact`, -claim the whole set it names atomically, and take the route it allows. For a -large source, divide independent sections among available agents while -retaining one owner for global coverage and dependency consistency. +run `autoform work impact`; under the open policy a retracted theorem stays an +open statement, so whatever rests on it stays conditionally proved. Retract only +that article and the dependents whose Markdown text the revision rewrites, +claiming them all in one acquire; the Lean-side impact decides every other +dependent. For a Lean revision requested in Human Review, record the decision in +the article and leave the Lean change to Formalize, which follows the [revision +contract](../../autoform_cli/README.md#revision-contract); this skill edits +only Markdown. For a large source, divide independent sections among available +agents while retaining one owner for global coverage and dependency +consistency. `open_statements` in `roadmap/README.md` is a project policy; absent means strict. `allowed` lets a theorem's statement land with a `sorry` proof, so @@ -91,6 +93,8 @@ dependents can be stated and proved before it is; the cost is conditionally proved results that stay incomplete until those proofs land. Change it only on the user's request and only once CI meets the [open statements](../../autoform_cli/README.md#open-statements) requirements. +Turn it back off only once no open statement remains: the strict audit rejects +every `sorry`, and strict status shows a proof resting on one as proved. Reconcile every affected source and milestone page, the coverage contract, `blueprint/README.md`, and the repository `README.md`. From bae97ab6050151164a217768d0534a45e0e59269 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 09:24:44 -0400 Subject: [PATCH 26/38] Run the impact probe's real-Lean test in CI tests/test_impact.py builds a Lean project and runs the work impact probe against it, and honours AUTOFORM_REQUIRE_REAL_LEAN_TESTS like the skeleton suite, but no CI job ran it with Lean installed, so it was always skipped. Add it to the real-Lean job. --- .github/workflows/tests.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index b0f055c7..cc66c8de 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -56,7 +56,7 @@ jobs: elan toolchain install "$(tr -d '\r\n' < tests/fixtures/skeleton-project/lean-toolchain)" command -v lake - name: Run the real-Lean suites - run: timeout --signal=TERM --kill-after=30s 25m uv run pytest -q tests/test_skeleton.py tests/test_lake_artifact_audit.py + run: timeout --signal=TERM --kill-after=30s 25m uv run pytest -q tests/test_skeleton.py tests/test_lake_artifact_audit.py tests/test_impact.py env: AUTOFORM_REQUIRE_REAL_LEAN_TESTS: "1" From 54089829c1221222b18bfa9fe837d0676b49e309 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 09:51:30 -0400 Subject: [PATCH 27/38] Read a partial def's body through its _unsafe_rec companion in work impact Lean adds a partial def to the kernel as an opaque constant whose value is only an Inhabited witness; the body it runs is the internal _unsafe_rec companion, which nothing else names. work impact therefore never reported an article whose lean: is a partial def that uses the revised set, so a revision could be judged contained while that definition stopped compiling. The probe now counts the companion as the opaque constant's value, as it counts an inductive's constructors. --- autoform_cli/probes/impact_probe.lean | 9 ++++++++- tests/test_impact.py | 8 +++++++- 2 files changed, 15 insertions(+), 2 deletions(-) diff --git a/autoform_cli/probes/impact_probe.lean b/autoform_cli/probes/impact_probe.lean index 8bf28cf2..889d7145 100644 --- a/autoform_cli/probes/impact_probe.lean +++ b/autoform_cli/probes/impact_probe.lean @@ -54,7 +54,9 @@ def namesJson (names : Array Name) : Json := /-- What a revision of any project-local constant can reach through `c`: the local constants its type and its value mention, and its deprecation state. Theorem values are read too (`allowOpaque`), since proofs break when the -statements they use change; an inductive's constructors stand in for a value. +statements they use change; an inductive's constructors stand in for a value, +and so does the `_unsafe_rec` companion of a `partial def`, whose kernel value +is only an `Inhabited` witness while the companion holds the body Lean runs. `internal` is judged on the user-facing name: a private declaration someone wrote is not an internal detail, while the companions Lean generates for it (`_proof_1`, `match_1`, `_simp_1`) still are. `user_name` is that name for a @@ -68,8 +70,13 @@ def record (env : Environment) (isLocal : Name → Bool) (c : Name) (info : Cons (module : Name) : Json := let value := info.value? (allowOpaque := true) let typeConstants := info.type.getUsedConstants + -- `Compiler.mkUnsafeRecName c`, spelled out to keep the probe's imports small. + let implementation := Name.str c "_unsafe_rec" let valueConstants := match info with | .inductInfo v => v.ctors.toArray + | .opaqueInfo _ => + let uses := (value.map (·.getUsedConstants)).getD #[] + if env.contains implementation then uses.push implementation else uses | _ => (value.map (·.getUsedConstants)).getD #[] let valueMissing := match info with | .thmInfo _ | .defnInfo _ | .opaqueInfo _ => value.isNone diff --git a/tests/test_impact.py b/tests/test_impact.py index c7b00d87..e0d87746 100644 --- a/tests/test_impact.py +++ b/tests/test_impact.py @@ -1170,6 +1170,8 @@ def seed : Nat := 2 theorem seedAgain : seed = 1 + 1 := seedEq +partial def seedLoop (n : Nat) : Nat := if n = 0 then seed else seedLoop (n - 1) + theorem usesAlias : seed = 2 ∧ True := ⟨seedAlias, trivial⟩ @[simp, deprecated seedEq (since := "2026-10-05")] @@ -1301,9 +1303,13 @@ def recorded(*args: object, **kwargs: object) -> str: ImpactArticle("chapter/seed-eq", None, ("Imp.seedEq",)), ImpactArticle("chapter/uses-alias", None, ("Imp.usesAlias",)), ImpactArticle("chapter/box", None, ("Imp.Box",)), + ImpactArticle("chapter/loop", None, ("Imp.seedLoop",)), ] seed = compute_impact(records, more, more[0], ["Imp.seed"], source_revision="rev") - assert _ids(seed.statement_impacted) == ["chapter/priv", "chapter/seed-eq", "chapter/uses-alias"] + # A partial def's kernel value is an `Inhabited` witness; its body is in + # the `_unsafe_rec` companion. + assert "Imp.seedLoop._unsafe_rec" in records["Imp.seedLoop"].value_uses + assert _ids(seed.statement_impacted) == ["chapter/loop", "chapter/priv", "chapter/seed-eq", "chapter/uses-alias"] assert [(helper.name, helper.owner) for helper in seed.helpers] == [ ("Imp.seedAgain", None), ("Imp.seedAlias", None), From b2d40edcc40c2cea555d3441828faa7a0a590265 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 09:56:12 -0400 Subject: [PATCH 28/38] Report a deprecated Lean target under doctor's lean targets check Doctor sorts audit findings into its "audit" and "lean targets" checks by code, and its list predates lean-target-deprecated, so a deprecated target failed the roadmap audit check while "lean targets" still said every target resolves. List the code with the other Lean target findings. --- autoform_cli/doctor.py | 1 + tests/test_doctor.py | 24 ++++++++++++++++++++++++ 2 files changed, 25 insertions(+) diff --git a/autoform_cli/doctor.py b/autoform_cli/doctor.py index 43b2440a..d855c160 100644 --- a/autoform_cli/doctor.py +++ b/autoform_cli/doctor.py @@ -28,6 +28,7 @@ _LEAN_FINDING_CODES = frozenset( { "invalid-lean-root", + "lean-target-deprecated", "lean-target-kind-mismatch", "lean-target-not-found", "missing-lean-target", diff --git a/tests/test_doctor.py b/tests/test_doctor.py index 508ab6a2..e9287468 100644 --- a/tests/test_doctor.py +++ b/tests/test_doctor.py @@ -166,6 +166,30 @@ def test_optional_lean_targets_report_success_missing_and_kind_mismatch(tmp_path assert _checks(mismatch)["lean targets"] == (False, "1 finding(s): lean-target-kind-mismatch") +def test_a_deprecated_lean_target_fails_the_lean_targets_check(tmp_path: Path) -> None: + project = _clean_project( + tmp_path, + metadata=( + "declaration: theorem", + "statement: formalized", + "proof: formalized", + "lean: Project.result", + ), + ) + lean_root = tmp_path / "lean" + lean_root.mkdir() + (lean_root / "Project.lean").write_text( + "theorem Project.fresh : True := trivial\n\n" + "@[deprecated Project.fresh] theorem Project.result : True := trivial\n", + encoding="utf-8", + ) + + result = diagnose_project(project, lean_root=lean_root) + + assert _checks(result)["lean targets"] == (False, "1 finding(s): lean-target-deprecated") + assert _checks(result)["audit"][0] + + def test_runtime_projection_failure_is_reported_without_traceback(tmp_path: Path) -> None: project = _clean_project( tmp_path, From 5648a8d5bb36b38c92c66c5ab9dc75908914644e Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 10:20:12 -0400 Subject: [PATCH 29/38] Run the open-statement audit and impact probe in their own real-Lean step The skeleton suite already uses most of its step's 25-minute timeout, so sharing that budget let a slow impact or audit test cut the skeleton suite short. Give the two suites a separate step with its own timeout, and leave the skeleton step as main has it. --- .github/workflows/tests.yml | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index cc66c8de..228a98ee 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -55,8 +55,12 @@ jobs: timeout --signal=TERM --kill-after=30s 10m \ elan toolchain install "$(tr -d '\r\n' < tests/fixtures/skeleton-project/lean-toolchain)" command -v lake - - name: Run the real-Lean suites - run: timeout --signal=TERM --kill-after=30s 25m uv run pytest -q tests/test_skeleton.py tests/test_lake_artifact_audit.py tests/test_impact.py + - name: Run the skeleton suite against real Lean + run: timeout --signal=TERM --kill-after=30s 25m uv run pytest -q tests/test_skeleton.py + env: + AUTOFORM_REQUIRE_REAL_LEAN_TESTS: "1" + - name: Run the open-statement audit and impact probe against real Lean + run: timeout --signal=TERM --kill-after=30s 10m uv run pytest -q tests/test_lake_artifact_audit.py tests/test_impact.py env: AUTOFORM_REQUIRE_REAL_LEAN_TESTS: "1" From f06277616528c3d35a4964e30b48b9e982cfbcd2 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 11:27:58 -0400 Subject: [PATCH 30/38] Say that a retracted theorem stays open until its proof is recorded The open-statements section said a retracted theorem stops being open once its statement is recorded again, but a restated theorem without a recorded proof is still an open statement: the audit keeps accepting its sorry and what rests on it stays conditional. Also say that the one-line limit per declaration covers the status lines, since errors are logged on their own. --- autoform_cli/README.md | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/autoform_cli/README.md b/autoform_cli/README.md index 468b052e..366f6a87 100644 --- a/autoform_cli/README.md +++ b/autoform_cli/README.md @@ -749,8 +749,8 @@ results that stay incomplete until every open statement they rest on is proved. A theorem that loses `statement` while its `lean:` still names a declaration, as a retraction or a revision leaves it, stays an open statement: that Lean, `sorry` or not, still compiles into whatever uses it. The audit keeps accepting -its `sorry`, and what rests on it stays conditional, until its statement is -recorded again or its `lean:` is removed. A definition is never open: its body +its `sorry`, and what rests on it stays conditional, until its proof is +recorded or its `lean:` is removed. A definition is never open: its body is its proof, CI rejects a `sorry` in it, and its statement phase waits until its proof prerequisites are stated. Turn the policy back off only once no open statement remains, since the strict audit rejects every `sorry` and strict @@ -785,9 +785,9 @@ that depends on `sorry`, an article declaration that reaches an open statement its Markdown dependencies do not reach (so a fully proved article, which assumes nothing, may reach none), a `lean:` name missing from the build, and an open statement that its article records as proved. Each article -declaration gets at most one log line, with `NAME` the declaration and `ID` the -article's node ID, and one with an error gets none of the last three, which -read as passing: +declaration gets at most one of these status lines, with `NAME` the +declaration and `ID` the article's node ID, and one with an error gets none +of the last three, which read as passing: ```text open statement (proof is sorry): NAME [ID] From 20eb5a427cba95465d8d22f365aac17522863525 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 12:06:44 -0400 Subject: [PATCH 31/38] Keep the conditional-proof render test off the Actions environment The test asserts the Lean row that the site shows for a declaration without a source link, but on GitHub Actions the renderer detects repository coordinates from GITHUB_REPOSITORY, GITHUB_SERVER_URL and GITHUB_SHA and links the source instead, so the row is absent and the test failed on CI only. Clear those variables and pass empty coordinates, as the stranded-links test already does. --- tests/test_render.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/tests/test_render.py b/tests/test_render.py index 7fe72c57..f7a3a8e9 100644 --- a/tests/test_render.py +++ b/tests/test_render.py @@ -1299,10 +1299,16 @@ def _conditional_project(tmp_path: Path, policy: str) -> Path: return project -def test_a_conditional_proof_names_the_open_statements_it_assumes(tmp_path: Path) -> None: +def test_a_conditional_proof_names_the_open_statements_it_assumes( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + # The Lean row is the fallback for a declaration without a source link, and + # on CI the Actions environment would supply the coordinates for one. + for variable in ("GITHUB_REPOSITORY", "GITHUB_SERVER_URL", "GITHUB_SHA"): + monkeypatch.delenv(variable, raising=False) project = _conditional_project(tmp_path, "allowed") - render_site(project / "blueprint", tmp_path / "out", lean_root=project) + render_site(project / "blueprint", tmp_path / "out", lean_root=project, repository_url="", ref="") page = (tmp_path / "out/roadmap/README.md").read_text(encoding="utf-8") top = page[page.index('id="top"'):] From 466e87eec0fb91f2423334ba72f1c448d1068875 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 14:15:16 -0400 Subject: [PATCH 32/38] Mark a retracted statement explicitly instead of inferring it from lean: A theorem with lean: but no statement: formalized used to count as an open statement, so a draft lean: name on a theorem that was never stated became an assumption its dependents could rest on, and CI accepted its sorry. Roadmap now records the retraction as statement: retracted, which the loader checks against lean:, proof: formalized and mathlib: true, and only stated or retracted theorems are open. The assumption contract also lists Mathlib articles that name lean: declarations, never open and allowed nothing, so CI can check those names exist and reach no open statement. Work items carry a revision flag that points Formalize at autoform work impact, and the assumptions text report calls an article conditional only when its derived state is, naming conditional articles without lean: too, and labels other listed articles that assume something as unproved. --- autoform_cli/__main__.py | 22 ++++++--- autoform_cli/graph.py | 24 +++++++++- autoform_cli/runtime.py | 3 ++ autoform_cli/status.py | 18 ++++---- autoform_cli/work.py | 20 ++++++--- tests/test_graph.py | 36 +++++++++++++++ tests/test_runtime.py | 25 +++++++++++ tests/test_status.py | 34 ++++++++++++-- tests/test_work.py | 97 ++++++++++++++++++++++++++++++++++++++-- 9 files changed, 254 insertions(+), 25 deletions(-) diff --git a/autoform_cli/__main__.py b/autoform_cli/__main__.py index 46f7b8dd..60aa4816 100644 --- a/autoform_cli/__main__.py +++ b/autoform_cli/__main__.py @@ -501,6 +501,8 @@ def _work(args: argparse.Namespace) -> int: print(_human_text(f"{item.phase}: {item.node_id}{durable} - {item.title}")) if item.assumes: print(_human_text(" assumes: " + ", ".join(item.assumes))) + if item.revision: + print(" revision: start from `autoform work impact`") return 0 if args.json: @@ -525,6 +527,8 @@ def _work(args: argparse.Namespace) -> int: print(_human_text(f"{item.title} ({item.node_id})")) print(f"State: {item.state}") print(f"Phase: {phase}") + if item.revision: + print("Revision: the statement was retracted; start from `autoform work impact`") if item.open_statements: print("Open statements: allowed") if item.assumes: @@ -549,6 +553,9 @@ def _work_assumptions(args: argparse.Namespace) -> int: # As in `_work`, only loading the roadmap is reported as a path error. try: contract = assumption_contract(args.target) + # The text report also names conditional articles without `lean:`, + # which the contract leaves out because CI has nothing to check there. + runtime = None if args.json else load_runtime_graph(args.target) except (GraphValidationError, RuntimeProjectionError) as error: for issue in error.issues: print(f"error: {_human_text(issue)}", file=sys.stderr) @@ -561,14 +568,19 @@ def _work_assumptions(args: argparse.Namespace) -> int: print(contract.to_json()) return 0 print(_human_text(f"Open statements: {'allowed' if contract.open_statements else 'forbidden'}")) - for article in contract.articles: - assumes = f"assumes {', '.join(article.assumes)}" - if article.open: + listed = {article.id: article for article in contract.articles} + for node in sorted(runtime.nodes, key=lambda candidate: candidate.id): + article = listed.get(node.id) + assumes = f"assumes {', '.join(node.status.assumes)}" + if article is not None and article.open: # An open statement is not conditional: its own proof is missing. line = f"open: {article.id} ({', '.join(article.declarations)})" print(_human_text(f"{line} {assumes}" if article.assumes else line)) - elif article.assumes: - print(_human_text(f"conditional: {article.id} {assumes}")) + elif node.status.state == "conditional": + print(_human_text(f"conditional: {node.id} {assumes}")) + elif article is not None and article.assumes: + # Not proved, so nothing is conditional yet; its proof would rest on these. + print(_human_text(f"unproved: {article.id} {assumes}")) return 0 diff --git a/autoform_cli/graph.py b/autoform_cli/graph.py index 916507a4..aa08c31c 100644 --- a/autoform_cli/graph.py +++ b/autoform_cli/graph.py @@ -39,6 +39,7 @@ } ) _FORMALIZED = "formalized" +_RETRACTED = "retracted" _TRUE = frozenset({"true", "yes"}) _FALSE = frozenset({"false", "no"}) @@ -78,6 +79,10 @@ class Node: lean: str | None = None declaration: str | None = None statement_formalized: bool = False + #: ``statement: retracted``: a revision retracted the statement while + #: ``lean:`` still names the old declaration, which stays in the build + #: until Formalize restates the article. + statement_retracted: bool = False proof_formalized: bool = False mathlib: bool = False mathlib_declaration: str | None = None @@ -227,6 +232,7 @@ def resolve(targets: tuple[str, ...], node: _ParsedNode = parsed_node) -> list[s declaration=metadata.get("declaration"), lean=metadata.get("lean"), statement_formalized=metadata.get("statement") == _FORMALIZED, + statement_retracted=metadata.get("statement") == _RETRACTED, proof_formalized=metadata.get("proof") == _FORMALIZED, mathlib=metadata.get("mathlib") in _TRUE, mathlib_declaration=metadata.get("mathlib_declaration"), @@ -472,6 +478,16 @@ def _parse_frontmatter(node_id: str, lines: list[str]) -> tuple[dict[str, str], continue metadata[key] = value + if metadata.get("statement") == _RETRACTED: + if "lean" not in metadata: + issues.append( + f"{node_id}: statement: retracted needs the lean: declaration it retracts;" + " without lean:, omit statement" + ) + if metadata.get("proof") == _FORMALIZED: + issues.append(f"{node_id}: proof: formalized needs statement: formalized, not retracted") + if metadata.get("mathlib") in _TRUE: + issues.append(f"{node_id}: a mathlib: true article cannot record statement: retracted") return metadata, end + 1, issues @@ -483,7 +499,13 @@ def _normalize_value(node_id: str, line_number: int, key: str, value: str) -> tu if not ARTICLE_ID_PATTERN.fullmatch(value): return value, f"{location}: malformed article_id {value!r}" return value, None - if key in {"statement", "proof"}: + if key == "statement": + if folded not in {_FORMALIZED, _RETRACTED}: + return value, ( + f"{location}: 'statement' accepts only {_FORMALIZED!r} or {_RETRACTED!r}; omit the key otherwise" + ) + return folded, None + if key == "proof": if folded != _FORMALIZED: return value, f"{location}: {key!r} accepts only {_FORMALIZED!r}; omit the key otherwise" return folded, None diff --git a/autoform_cli/runtime.py b/autoform_cli/runtime.py index 49db6b8f..d3cb1250 100644 --- a/autoform_cli/runtime.py +++ b/autoform_cli/runtime.py @@ -42,6 +42,7 @@ class RuntimeAssertions: """Authored facts copied from article frontmatter.""" statement_formalized: bool + statement_retracted: bool proof_formalized: bool not_ready: bool @@ -50,6 +51,7 @@ def as_dict(self) -> dict[str, bool]: "not_ready": self.not_ready, "proof_formalized": self.proof_formalized, "statement_formalized": self.statement_formalized, + "statement_retracted": self.statement_retracted, } @@ -347,6 +349,7 @@ def build_runtime_graph( dependencies=node.dependencies, assertions=RuntimeAssertions( statement_formalized=node.statement_formalized, + statement_retracted=node.statement_retracted, proof_formalized=node.proof_formalized, not_ready=node.not_ready, ), diff --git a/autoform_cli/status.py b/autoform_cli/status.py index 77081565..3742ecbc 100644 --- a/autoform_cli/status.py +++ b/autoform_cli/status.py @@ -104,9 +104,9 @@ class NodeStatus: ``waiting_on`` names the prerequisites that keep an unproved node from its next phase, in authored order. ``assumes`` names the open statements (not - proved, but stated or a theorem still naming its ``lean:`` declaration) a - proof of this node rests on; it is empty unless the project allows open - statements. + proved, but stated or a theorem whose ``statement: retracted`` keeps its + ``lean:`` declaration in the build) a proof of this node rests on; it is + empty unless the project allows open statements. """ node_id: str @@ -181,16 +181,18 @@ def done(dependency_id: str, attribute: str) -> bool: can_prove = stated and not unstated and not unmet waiting = unstated + unmet if stated or definition else unstated # Mathlib and unstated nodes reach nothing, so they get no entry, - # except an article whose statement a revision retracted while its - # `lean:` still names the old declaration: that declaration and its - # sorry stay in the build until Formalize restates it. A retracted - # definition is not open, but its body still reaches what it used. + # except an article recording `statement: retracted`: a revision + # retracted its statement while its `lean:` still names the old + # declaration, which stays in the build, sorry and all, until + # Formalize restates it. A draft `lean:` on a never-stated theorem + # is not an assumption. A retracted definition is not open, but its + # body still reaches what it used. if not node.mathlib: reached = frozenset().union(*(reaches.get(other, frozenset()) for other in node.dependencies)) assumes = tuple(sorted(reached)) if proved or (definition and node.lean): reaches[node_id] = reached - elif stated or node.lean: + elif stated or node.statement_retracted: reaches[node_id] = frozenset({node_id}).union( *(reaches.get(other, frozenset()) for other in node.statement_dependencies) ) diff --git a/autoform_cli/work.py b/autoform_cli/work.py index ada4b925..89661fd4 100644 --- a/autoform_cli/work.py +++ b/autoform_cli/work.py @@ -46,6 +46,9 @@ class WorkItem: lean_targets: tuple[WorkLeanTarget, ...] assumes: tuple[str, ...] = () open_statements: bool = False + #: The article records ``statement: retracted``: a revision retracted its + #: statement, so the work starts from ``autoform work impact``. + revision: bool = False @property def ready(self) -> bool: @@ -65,6 +68,7 @@ def as_dict(self) -> dict[str, object]: "open_statements": self.open_statements, "phase": self.phase, "ready": self.ready, + "revision": self.revision, "source_targets": list(self.source_targets), "state": self.state, "title": self.title, @@ -137,6 +141,7 @@ def _item(node: RuntimeNode, *, open_statements: bool) -> WorkItem: ), assumes=node.status.assumes, open_statements=open_statements, + revision=node.assertions.statement_retracted, ) @@ -252,9 +257,13 @@ def assumption_contract(project_or_blueprint: str | Path) -> AssumptionContract: CI audits the Lean build against this: an open article's own declarations may keep a ``sorry`` proof, and every other article may reach only the open - statements its Markdown dependencies declare. A theorem whose statement a - revision retracted stays open while its ``lean:`` names the old declaration. - Under the strict policy no article is open and nothing is allowed. + statements its Markdown dependencies declare. A theorem recording + ``statement: retracted`` stays open while its ``lean:`` names the old + declaration; a theorem that was never stated is not open, even with a + ``lean:`` name, so CI rejects its ``sorry``. Mathlib articles are listed too, + never open and allowed nothing, so CI can check that their names exist and + reach no open statement. Under the strict policy no article is open and + nothing is allowed. """ runtime = load_runtime_graph(project_or_blueprint) declarations = { @@ -263,12 +272,13 @@ def assumption_contract(project_or_blueprint: str | Path) -> AssumptionContract: } articles: list[AssumptionArticle] = [] for node in sorted(runtime.nodes, key=lambda candidate: candidate.id): - if node.mathlib or not declarations[node.id]: + if not declarations[node.id]: continue is_open = ( runtime.open_statements and not node.status.proved - and (node.status.stated or not is_definition(node)) + and not is_definition(node) + and (node.status.stated or node.assertions.statement_retracted) ) allowed = {name for assumed in node.status.assumes for name in declarations.get(assumed, ())} if is_open: diff --git a/tests/test_graph.py b/tests/test_graph.py index 7d838a41..7a7da9fe 100644 --- a/tests/test_graph.py +++ b/tests/test_graph.py @@ -279,6 +279,7 @@ def test_splits_statement_and_proof_dependencies(tmp_path: Path) -> None: ("metadata", "message"), [ ({"statement": "yes"}, "accepts only 'formalized'"), + ({"statement": "bogus"}, "'statement' accepts only 'formalized' or 'retracted'"), ({"proof": "sorry"}, "accepts only 'formalized'"), ({"mathlib": "maybe"}, "accepts only true or false"), ({"not_ready": "1"}, "accepts only true or false"), @@ -293,6 +294,41 @@ def test_rejects_invalid_assertions(tmp_path: Path, metadata: dict[str, str], me load_graph(blueprint) +def test_records_a_retracted_statement(tmp_path: Path) -> None: + blueprint = tmp_path / "blueprint" + _node(blueprint, "result.md", "# Result\n", declaration="theorem", statement="Retracted", lean="Ns.result") + + node = load_graph(blueprint).nodes["result"] + + assert (node.statement_retracted, node.statement_formalized, node.proof_formalized) == (True, False, False) + + +@pytest.mark.parametrize( + ("metadata", "message"), + [ + ({}, "result: statement: retracted needs the lean: declaration it retracts; without lean:, omit statement"), + ( + {"lean": "Ns.result", "proof": "formalized"}, + "result: proof: formalized needs statement: formalized, not retracted", + ), + ( + {"lean": "Ns.result", "mathlib": "true"}, + "result: a mathlib: true article cannot record statement: retracted", + ), + ], +) +def test_rejects_a_retracted_statement_that_cannot_keep_its_declaration( + tmp_path: Path, metadata: dict[str, str], message: str +) -> None: + blueprint = tmp_path / "blueprint" + _node(blueprint, "result.md", "# Result\n", declaration="theorem", statement="retracted", **metadata) + + with pytest.raises(GraphValidationError) as raised: + load_graph(blueprint) + + assert raised.value.issues == (message,) + + def test_records_origin_and_source_links_without_treating_them_as_edges(tmp_path: Path) -> None: blueprint = tmp_path / "blueprint" source = blueprint / "sources" / "paper.md" diff --git a/tests/test_runtime.py b/tests/test_runtime.py index f67c09ae..35fd2d7d 100644 --- a/tests/test_runtime.py +++ b/tests/test_runtime.py @@ -342,3 +342,28 @@ def test_open_runtime_records_the_policy_and_what_each_proof_assumes(tmp_path: P assert (waiting["assumes"], waiting["waiting_on"]) == ([], ["chapter/section/gap"]) assert statuses["open"]["state"] == "can_prove" assert statuses["open"]["assumes"] == [] + + +def test_runtime_assertions_record_a_retracted_statement(tmp_path: Path) -> None: + project = _policy_project(tmp_path, "allowed") + _article( + project, + "chapter/section/open.md", + title="Open", + declaration="theorem", + statement="retracted", + lean="Project.open_thm", + ) + + payload = json.loads(load_runtime_graph(project).to_json()) + nodes = {node["id"].removeprefix("chapter/section/"): node for node in payload["nodes"]} + + assert nodes["open"]["assertions"] == { + "not_ready": False, + "proof_formalized": False, + "statement_formalized": False, + "statement_retracted": True, + } + assert nodes["reduction"]["assertions"]["statement_retracted"] is False + # The old declaration stays in the build, so the reduction still rests on it. + assert nodes["reduction"]["status"]["assumes"] == ["chapter/section/open"] diff --git a/tests/test_status.py b/tests/test_status.py index fb4af63b..8112ac53 100644 --- a/tests/test_status.py +++ b/tests/test_status.py @@ -349,8 +349,8 @@ def test_a_retracted_theorem_still_naming_its_lean_stays_an_open_statement(tmp_p """ blueprint = tmp_path / "blueprint" _node(blueprint, "README.md", open_statements=policy) - _node(blueprint, "retracted.md", declaration="theorem", lean="Ns.retracted") - _node(blueprint, "old.md", declaration="def", lean="Ns.old") + _node(blueprint, "retracted.md", declaration="theorem", statement="retracted", lean="Ns.retracted") + _node(blueprint, "old.md", declaration="def", statement="retracted", lean="Ns.old") _node(blueprint, "gone.md", declaration="theorem") _node( blueprint, @@ -377,7 +377,14 @@ def test_a_retracted_definition_still_passes_on_what_its_body_reaches(tmp_path: blueprint = tmp_path / "blueprint" _node(blueprint, "README.md", open_statements=policy) _node(blueprint, "open.md", declaration="theorem", statement="formalized", lean="Ns.open") - _node(blueprint, "old.md", "## Proof depends on\n\n- [Open](open.md)\n", declaration="def", lean="Ns.old") + _node( + blueprint, + "old.md", + "## Proof depends on\n\n- [Open](open.md)\n", + declaration="def", + statement="retracted", + lean="Ns.old", + ) _node( blueprint, "uses.md", @@ -396,6 +403,27 @@ def test_a_retracted_definition_still_passes_on_what_its_body_reaches(tmp_path: assert (statuses["uses"].key, statuses["uses"].assumes) == ("proved", ()) +def test_a_never_stated_theorem_naming_a_draft_lean_is_not_an_open_statement(tmp_path: Path) -> None: + """A draft ``lean:`` name is not an assumption: only a stated or retracted theorem is open.""" + blueprint = tmp_path / "blueprint" + _node(blueprint, "README.md", open_statements="allowed") + _node(blueprint, "draft.md", declaration="theorem", lean="Ns.draft") + _node( + blueprint, + "uses.md", + "## Proof depends on\n\n- [Draft](draft.md)\n", + declaration="theorem", + statement="formalized", + proof="formalized", + lean="Ns.uses", + ) + + statuses = derive(load_graph(blueprint)) + + assert statuses["draft"].key == "can_state" + assert (statuses["uses"].key, statuses["uses"].assumes) == ("proved", ()) + + @pytest.mark.parametrize("open_statements", [False, True]) def test_a_missing_dependency_counts_as_neither_stated_nor_proved( tmp_path: Path, open_statements: bool diff --git a/tests/test_work.py b/tests/test_work.py index 918ea282..70f766b5 100644 --- a/tests/test_work.py +++ b/tests/test_work.py @@ -343,6 +343,7 @@ def test_work_cli_emits_stable_json(tmp_path: Path, capsys) -> None: "open_statements": False, "phase": "proof", "ready": True, + "revision": False, "source_targets": [], "state": "can_prove", "title": "Prove me", @@ -528,7 +529,7 @@ def _policy_project(tmp_path: Path, policy: str | None) -> Path: metadata=["article_id: af_00000000000000000000000e", "declaration: theorem"], depends="reduction.md", ) - # Neither belongs in the assumption contract: one is upstream, the other names no declaration. + # The contract lists the upstream article, never open, but not the one that names no declaration. _article( project, "upstream.md", @@ -731,7 +732,7 @@ def test_open_work_list_names_the_policy_even_when_nothing_is_ready(tmp_path: Pa def _contract_article( node_id: str, - article_id: str, + article_id: str | None, state: str, declarations: list[str], *, @@ -770,6 +771,7 @@ def test_work_assumptions_under_the_strict_policy_lists_articles_with_nothing_op ), _contract_article("chapter/prove", "af_000000000000000000000003", "can_prove", ["Project.prove"]), _contract_article("chapter/reduction", "af_00000000000000000000000c", "proved", ["Project.reduction"]), + _contract_article("chapter/upstream", None, "mathlib", ["Project.upstream"]), _contract_article("chapter/uses", "af_00000000000000000000000d", "can_prove", ["Project.uses"]), ], } @@ -817,6 +819,7 @@ def test_work_assumptions_under_the_open_policy_bounds_each_article( assumes=("chapter/open",), allowed=open_declarations, ), + _contract_article("chapter/upstream", None, "mathlib", ["Project.upstream"]), _contract_article( "chapter/uses", "af_00000000000000000000000d", @@ -849,7 +852,7 @@ def test_work_assumptions_keeps_a_retracted_theorem_while_its_lean_names_the_old ) -> None: """Roadmap retracts a statement but keeps `lean:`; the old sorry stays declared until Formalize restates it.""" project = _policy_project(tmp_path, policy) - _edit(project, "open.md", "statement: formalized\n", "") + _edit(project, "open.md", "statement: formalized\n", "statement: retracted\n") allowed = ("Project.open_aux", "Project.open_thm") if policy == "allowed" else () assert cli.main(["work", "assumptions", str(project), "--json"]) == 0 @@ -889,6 +892,94 @@ def test_work_assumptions_bounds_a_proof_recorded_without_its_statement( ) +def test_work_assumptions_does_not_open_a_never_stated_theorem_naming_a_draft_lean(tmp_path: Path, capsys) -> None: + """A draft `lean:` name on a theorem that was never stated is no assumption, so CI rejects its sorry.""" + project = _policy_project(tmp_path, "allowed") + _edit(project, "open.md", "statement: formalized\n", "") + + assert cli.main(["work", "assumptions", str(project), "--json"]) == 0 + articles = {article["id"]: article for article in json.loads(capsys.readouterr().out)["articles"]} + + assert articles["chapter/open"] == _contract_article( + "chapter/open", "af_00000000000000000000000b", "can_state", ["Project.open_thm", "Project.open_aux"] + ) + assert articles["chapter/reduction"] == _contract_article( + "chapter/reduction", "af_00000000000000000000000c", "proved", ["Project.reduction"] + ) + assert not any( + name in article["allowed_open_declarations"] + for article in articles.values() + for name in ("Project.open_thm", "Project.open_aux") + ) + + +def test_work_assumptions_text_labels_only_conditional_articles_as_conditional(tmp_path: Path, capsys) -> None: + """A conditional article without `lean:` is named too; an unproved one that assumes something is not conditional.""" + project = _policy_project(tmp_path, "allowed") + _article( + project, + "bare.md", + title="Bare", + metadata=["declaration: theorem", "statement: formalized", "proof: formalized"], + proof_depends="open.md", + ) + _article( + project, + "old.md", + title="Old", + metadata=["declaration: def", "statement: retracted", "lean: Project.old"], + proof_depends="open.md", + ) + + assert cli.main(["work", "assumptions", str(project), "--json"]) == 0 + articles = {article["id"]: article for article in json.loads(capsys.readouterr().out)["articles"]} + assert "chapter/bare" not in articles + assert articles["chapter/old"] == _contract_article( + "chapter/old", + None, + "can_state", + ["Project.old"], + assumes=("chapter/open",), + allowed=("Project.open_aux", "Project.open_thm"), + ) + + # chapter/corollary assumes chapter/open too, but names no declaration and is not proved. + assert cli.main(["work", "assumptions", str(project)]) == 0 + assert capsys.readouterr().out == ( + "Open statements: allowed\n" + "conditional: chapter/bare assumes chapter/open\n" + "unproved: chapter/old assumes chapter/open\n" + "open: chapter/open (Project.open_thm, Project.open_aux)\n" + "open: chapter/prove (Project.prove)\n" + "conditional: chapter/reduction assumes chapter/open\n" + "open: chapter/uses (Project.uses) assumes chapter/open\n" + ) + + +def test_work_flags_a_retracted_article_as_a_revision(tmp_path: Path, capsys) -> None: + project = _policy_project(tmp_path, "allowed") + _edit(project, "open.md", "statement: formalized\n", "statement: retracted\n") + + assert cli.main(["work", "list", str(project), "--json"]) == 0 + items = {item["node_id"]: item for item in json.loads(capsys.readouterr().out)["items"]} + assert (items["chapter/open"]["phase"], items["chapter/open"]["revision"]) == ("statement", True) + assert {node_id for node_id, item in items.items() if item["revision"]} == {"chapter/open"} + + assert cli.main(["work", "list", str(project)]) == 0 + lines = capsys.readouterr().out.splitlines() + revision = " revision: start from `autoform work impact`" + opened = lines.index("statement: chapter/open [af_00000000000000000000000b] - Open") + assert lines[opened + 1] == revision + assert lines.count(revision) == 1 + + for selector, flagged in (("chapter/open", True), ("chapter/prove", False)): + assert cli.main(["work", "context", selector, str(project), "--json"]) == 0 + assert json.loads(capsys.readouterr().out)["item"]["revision"] is flagged + assert cli.main(["work", "context", selector, str(project)]) == 0 + context = capsys.readouterr().out.splitlines() + assert ("Revision: the statement was retracted; start from `autoform work impact`" in context) is flagged + + def test_work_assumptions_reports_errors_on_stderr_with_exit_2( tmp_path: Path, capsys, monkeypatch: pytest.MonkeyPatch ) -> None: From 019ed2eb66e0ed8452c9af4bc4e8633ceb231c46 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 14:10:44 -0400 Subject: [PATCH 33/38] Give every impacted helper a claim target, keyed by its own name when no article owns it A helper no article owns was listed in work impact but claimed by nobody, so two revisions that both had to repair it could proceed at once without contending. Each helper now carries claim_target: its owner's claim target, or a lean/ key derived from its name the way author claim keys are, and claim_targets includes every helper's. Revisions touching the same unowned helper therefore contend for one claim. The JSON helper objects gain claim_target and the text report prints it; contained is unchanged, since an unowned helper already made a revision not contained. --- autoform_cli/impact.py | 27 +++++++++++--- tests/test_impact.py | 83 +++++++++++++++++++++++++++++++++++++----- 2 files changed, 94 insertions(+), 16 deletions(-) diff --git a/autoform_cli/impact.py b/autoform_cli/impact.py index 08a7abaa..715fa9bb 100644 --- a/autoform_cli/impact.py +++ b/autoform_cli/impact.py @@ -11,6 +11,7 @@ from __future__ import annotations +import hashlib import json import os import re @@ -335,6 +336,9 @@ class ImpactHelper: path: str | None line: int | None owner: str | None + #: The owner's claim target, else a key of the helper's own, so that + #: revisions touching the same unowned helper contend for one claim. + claim_target: str def as_dict(self) -> dict[str, object]: return { @@ -345,6 +349,7 @@ def as_dict(self) -> dict[str, object]: "path": self.path, "line": self.line, "owner": self.owner, + "claim_target": self.claim_target, } @@ -505,6 +510,7 @@ def compute_impact( statement_impacted.sort(key=lambda item: item.id) proof_impacted.sort(key=lambda item: item.id) + by_id = {article.id: article for article in articles} helpers: list[ImpactHelper] = [] for name in sorted(meaning | proof): record = records[name] @@ -512,9 +518,10 @@ def compute_impact( continue path, line = locate(record) if locate is not None else (None, None) impact = "statement" if name in meaning else "proof" - helpers.append(ImpactHelper(name, record.kind, impact, record.module, path, line, _owner(record, records, named))) + owner = _owner(record, records, named) + target = by_id[owner].claim_target if owner in by_id else _helper_claim_key(name) + helpers.append(ImpactHelper(name, record.kind, impact, record.module, path, line, owner, target)) - by_id = {article.id: article for article in articles} impacted = (*statement_impacted, *proof_impacted) undeclared = sorted(item.id for item in impacted if not _reaches(item.id, revised.id, by_id)) @@ -533,9 +540,9 @@ def compute_impact( if record.deprecated ) - # A helper is repaired under its owner's claim, so the owner is claimed too. - owners = {by_id[helper.owner].claim_target for helper in helpers if helper.owner in by_id} - others = {item.claim_target for item in impacted} | owners + # A helper is repaired under its owner's claim, so the owner is claimed + # too; an unowned helper is repaired under a claim keyed by its own name. + others = {item.claim_target for item in impacted} | {helper.claim_target for helper in helpers} claim_targets = (revised.claim_target, *sorted(others - {revised.claim_target})) return ImpactReport( source_revision=source_revision, @@ -551,6 +558,14 @@ def compute_impact( ) +def _helper_claim_key(name: str) -> str: + """The claim key of a helper no article owns, shaped like ``claims.author_claim_key``.""" + + slug = re.sub(r"[^a-z0-9-]+", "-", name.lower()).strip("-")[:48] or "declaration" + digest = hashlib.sha256(name.encode("utf-8")).hexdigest()[:16] + return f"lean/{slug}-{digest}" + + def _resolver(records: Mapping[str, ConstantRecord]) -> Callable[..., str | None]: """Resolve a user-facing name to a record: its exact name, else the private constant it names. @@ -773,7 +788,7 @@ def format_impact(report: ImpactReport) -> list[str]: if helper.path and helper.line is not None: where = f"{where}:{helper.line}" owner = f"owner {helper.owner}" if helper.owner else "no owner" - lines.append(f" {helper.name} ({helper.kind}, {helper.impact}) {where}; {owner}") + lines.append(f" {helper.name} ({helper.kind}, {helper.impact}) {where}; {owner}; claim {helper.claim_target}") if report.undeclared_dependencies: lines.append( "Impacted without a Markdown dependency path to the revised article: " diff --git a/tests/test_impact.py b/tests/test_impact.py index bd20cacf..41607bcf 100644 --- a/tests/test_impact.py +++ b/tests/test_impact.py @@ -1,14 +1,16 @@ from __future__ import annotations +import hashlib import json import os +import re import shutil import subprocess from pathlib import Path import pytest -from autoform_cli import __main__ as cli, skeleton +from autoform_cli import __main__ as cli, claims, skeleton from autoform_cli.impact import ( IMPACT_MARKER, IMPACT_SCHEMA, @@ -66,6 +68,11 @@ def _ids(items) -> list[str]: return [item.id for item in items] +def _lean_key(name: str) -> str: + slug = re.sub(r"[^a-z0-9-]+", "-", name.lower()).strip("-")[:48] or "declaration" + return f"lean/{slug}-{hashlib.sha256(name.encode()).hexdigest()[:16]}" + + def test_type_uses_impact_statements_and_theorem_values_impact_proofs() -> None: records = _records( _rec("A.f", "def"), @@ -191,6 +198,7 @@ def locate(record: ConstantRecord) -> tuple[str | None, int | None]: "path": "Demo.lean", "line": 1, "owner": "shared-a", + "claim_target": "shared-a", }, { "name": "A.uses.aux", @@ -200,6 +208,7 @@ def locate(record: ConstantRecord) -> tuple[str | None, int | None]: "path": "Demo.lean", "line": 2, "owner": "uses", + "claim_target": "uses", }, { "name": "A.uses.aux.deep", @@ -209,6 +218,7 @@ def locate(record: ConstantRecord) -> tuple[str | None, int | None]: "path": "Demo.lean", "line": 3, "owner": "uses", + "claim_target": "uses", }, { "name": "_private.Demo.Extra.0.A.priv", @@ -218,12 +228,14 @@ def locate(record: ConstantRecord) -> tuple[str | None, int | None]: "path": None, "line": None, "owner": None, + "claim_target": _lean_key("_private.Demo.Extra.0.A.priv"), }, ] assert located == [helper.name for helper in report.helpers] assert _ids(report.statement_impacted) == ["uses"] - # The helpers are repaired under their owners' claims, so shared-a is claimed too. - assert report.claim_targets == ("base", "shared-a", "uses") + # The helpers are repaired under their owners' claims, so shared-a is + # claimed too, and the unowned helper under a key of its own. + assert report.claim_targets == ("base", _lean_key("_private.Demo.Extra.0.A.priv"), "shared-a", "uses") def test_revised_names_resolve_by_component_and_are_never_their_own_helpers() -> None: @@ -401,7 +413,7 @@ def test_an_alias_shares_its_target_s_statement() -> None: assert _ids(report.statement_impacted) == ["states-alias"] assert _ids(report.proof_impacted) == ["uses-alias"] assert [(helper.name, helper.impact) for helper in report.helpers] == [("A.alias", "statement")] - assert report.claim_targets == ("plain", "states-alias", "uses-alias") + assert report.claim_targets == ("plain", _lean_key("A.alias"), "states-alias", "uses-alias") def test_a_deprecated_constant_s_own_companions_are_not_its_users() -> None: @@ -484,7 +496,54 @@ def test_helpers_the_revised_article_owns_keep_a_revision_contained() -> None: assert format_impact(owned)[2] == "Contained: nothing outside s uses A.S, so it can be revised in place." assert not shared.contained assert [(helper.name, helper.owner) for helper in shared.helpers] == [("A.T.aux", "other"), ("A.loose", None)] - assert shared.claim_targets == ("t", "other") + assert shared.claim_targets == ("t", _lean_key("A.loose"), "other") + + +def test_revisions_touching_one_unowned_helper_contend_for_its_claim() -> None: + records = _records( + _rec("A.left", "def"), + _rec("A.right", "def"), + _rec("A.bridge", type_uses=("A.left", "A.right")), + _rec("A.T", "inductive"), + _rec("A.T.aux", type_uses=("A.left",), parent="A.T"), + ) + articles = [ + _article("left", "A.left", article_id="af_left"), + _article("right", "A.right", article_id="af_right"), + _article("t", "A.T", article_id="af_t"), + ] + + left = _impact(records, articles, "left") + right = _impact(records, articles, "right") + + # No article names A.bridge, so both revisions claim one key derived from + # its name, shaped like an author claim key; an owned helper is claimed + # under its owner's claim target. + key = "lean/a-bridge-" + hashlib.sha256(b"A.bridge").hexdigest()[:16] + assert claims._validate_key(key) == key + assert [(helper.name, helper.owner, helper.claim_target) for helper in left.helpers] == [ + ("A.T.aux", "t", "af_t"), + ("A.bridge", None, key), + ] + assert [(helper.name, helper.claim_target) for helper in right.helpers] == [("A.bridge", key)] + assert left.claim_targets == ("af_left", "af_t", key) + assert right.claim_targets == ("af_right", key) + assert not right.contained + assert json.loads(right.to_json())["helpers"][0]["claim_target"] == key + + +def test_an_unowned_helper_claim_key_is_ref_safe_for_any_name() -> None: + names = ("_private.Demo.Extra.0.A.priv", "A.«weird name»", "«∀»", "A." + "long" * 20) + records = _records(_rec("A.base", "def"), *(_rec(name, type_uses=("A.base",)) for name in names)) + articles = [_article("base", "A.base")] + + report = _impact(records, articles, "base") + + keys = {helper.name: helper.claim_target for helper in report.helpers} + assert keys == {name: _lean_key(name) for name in names} + assert keys["«∀»"].startswith("lean/declaration-") + assert all(claims._validate_key(key) == key for key in keys.values()) + assert len(keys["A." + "long" * 20]) == len("lean/") + 48 + 1 + 16 # --------------------------------------------------------------------------- # @@ -836,6 +895,7 @@ def test_cli_writes_the_impact_report_as_canonical_json(tmp_path: Path, monkeypa "path": "Demo.lean", "line": 5, "owner": None, + "claim_target": "lean/demo-base-eq-7f17aa41d1461243", } ], "undeclared_dependencies": ["chapter/loose"], @@ -844,7 +904,7 @@ def test_cli_writes_the_impact_report_as_canonical_json(tmp_path: Path, monkeypa {"name": "Demo.old", "replacement": "Demo.base_eq", "users": ["Demo.loose"], "articles": []}, ], "deprecated_unused": ["Demo.gone"], - "claim_targets": [_BASE_ID, _USES_ID, "chapter/loose"], + "claim_targets": [_BASE_ID, _USES_ID, "chapter/loose", "lean/demo-base-eq-7f17aa41d1461243"], } (call,) = calls assert call["label"] == "impact probe" @@ -871,12 +931,12 @@ def test_cli_text_report_lists_each_section(tmp_path: Path, monkeypatch, capsys) "Proof impacted:", " chapter/loose: Demo.loose", "Helpers no article names:", - " Demo.base_eq (theorem, statement) Demo.lean:5; no owner", + " Demo.base_eq (theorem, statement) Demo.lean:5; no owner; claim lean/demo-base-eq-7f17aa41d1461243", "Impacted without a Markdown dependency path to the revised article: chapter/loose", "Deprecated:", " Demo.gone: no users, safe to delete", " Demo.old -> Demo.base_eq: used by Demo.loose", - f"Claim targets: {_BASE_ID}, {_USES_ID}, chapter/loose", + f"Claim targets: {_BASE_ID}, {_USES_ID}, chapter/loose, lean/demo-base-eq-7f17aa41d1461243", ] assert calls[0]["timeout"] == skeleton.DEFAULT_PROBE_TIMEOUT @@ -921,7 +981,8 @@ def test_cli_text_escapes_terminal_control_characters(tmp_path: Path, monkeypatc output = capsys.readouterr() assert code == 0, output.err assert "\x1b" not in output.out - assert " Demo.bad\\x1b[2Jname (theorem, statement) Demo.lean; no owner" in output.out.splitlines() + line = " Demo.bad\\x1b[2Jname (theorem, statement) Demo.lean; no owner; claim lean/demo-bad-2jname-6d30903d669991fb" + assert line in output.out.splitlines() @pytest.mark.parametrize( @@ -1265,7 +1326,9 @@ def recorded(*args: object, **kwargs: object) -> str: {"name": "Imp.oldUnused", "replacement": "Imp.base_eq", "users": [], "articles": []}, ] assert report["deprecated_unused"] == ["Imp.oldSeed", "Imp.oldUnused"] - assert report["claim_targets"] == ["chapter/base", "chapter/proved", "chapter/simp", "chapter/uses"] + unowned = sorted(_lean_key(h["name"]) for h in report["helpers"]) + assert [h["claim_target"] for h in report["helpers"]] == [_lean_key(h["name"]) for h in report["helpers"]] + assert report["claim_targets"] == ["chapter/base", "chapter/proved", "chapter/simp", "chapter/uses", *unowned] assert report["source_revision"] == source_revision (probed,) = outputs From 0da3cac046b3e81684631681599a17dd5cc36b5b Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 14:11:20 -0400 Subject: [PATCH 34/38] Document the retraction marker and close the revision contract's gaps A Lean revision requested in Human Review was stranded: Roadmap only recorded the decision, so the article stayed proved and work list never offered it. Roadmap now retracts the article with statement: retracted, keeping lean:, so it returns to the frontier as a statement phase flagged as a revision. The README and skills now describe the marker and which articles are open statements, the mathlib articles in the assumptions contract, the revision flag on work items, and the lean/- claim key for unowned helpers. The revision contract defines a claim set per route, migrates proof-impacted articles off a sorry'd X under the expand route (otherwise X never becomes unused and R can never record its proof), re-runs work impact after the rebase, and warns that instance, attribute, and notation changes are invisible to it. Several statements that contradicted the code are corrected, and the audit's sorry-free hint now says to restate a retracted article before recording its proof. --- autoform_cli/README.md | 169 ++++++++++++------ .../templates/github/autoform_audit.py | 2 +- skills/formalize/SKILL.md | 18 +- skills/human-review/SKILL.md | 6 +- skills/roadmap/SKILL.md | 16 +- .../.github/autoform_audit.py | 2 +- 6 files changed, 137 insertions(+), 76 deletions(-) diff --git a/autoform_cli/README.md b/autoform_cli/README.md index 7b7aa237..3cebdb6d 100644 --- a/autoform_cli/README.md +++ b/autoform_cli/README.md @@ -84,7 +84,8 @@ An article asserts only facts a human or agent verified: | Key | Meaning | | --- | --- | | `statement: formalized` | The Lean statement exists and compiles. | -| `proof: formalized` | The Lean proof is complete. | +| `statement: retracted` | A revision retracted the statement while `lean:` still names the old declaration, which stays in the build until Formalize restates the article and records `statement: formalized` in its place. Requires `lean:`; invalid with `proof: formalized` or `mathlib: true`. | +| `proof: formalized` | The Lean proof compiles. Under the open policy it may rest on open statements, and the article is then conditional; only the derived `fully_proved` means complete and `sorry`-free. | | `mathlib: true` | The result is upstreamed into Mathlib. | | `not_ready: true` | Needs more blueprint work before it can be attempted. | | `lean: Ns.decl` | Declaration name(s) that discharge the article. | @@ -102,19 +103,19 @@ lands only with its proof; under `open_statements: allowed` it may land with a | --- | --- | | `can_state` | Every statement prerequisite is stated and every proof prerequisite is proved. Under the open policy a theorem waits for no proof prerequisite, and a definition, whose body is its proof, waits for them to be stated. | | `can_prove` | Stated, every statement prerequisite is stated, and every proof prerequisite is proved (strict policy) or stated (open policy). | -| `proved` | The proof compiles. | +| `proved` | The article records `proof: formalized`, is a stated definition, or is in Mathlib. | | `conditional` | Proved, but the proof rests on an open statement; open policy only. | | `fully_proved` | Proved, and every prerequisite is fully proved, recursively. | -| `defined` | A definition is written but rests on unfinished work. | +| `defined` | A definition is written but rests on unfinished work; one whose body rests on an open statement is `conditional` instead. | `proved` and `fully_proved` differ on purpose: a theorem whose own proof compiles but which rests on unfinished work is green, not dark green. `conditional`, labelled "conditionally proved", is the open policy's case of -that: the article records `proof: formalized` and its proof compiles, but it -reaches an open statement, an article whose statement is formalized and whose -proof is not, or a retracted theorem (see [Open statements](#open-statements)), -through its dependencies. An open dependency counts together with -whatever its statement prerequisites reach, and a proved dependency passes on +that: the article is proved, but it reaches an open statement, a theorem whose +`lean:` names a declaration and which is stated or retracted but not proved +(see [Open statements](#open-statements)), through its dependencies. An open +dependency counts together with whatever its statement prerequisites reach, and +a proved dependency passes on everything it reaches. The site colours it violet, never green, and lists those open statements in an `Assumes` row on the article page. It is never `fully_proved`, which keeps its meaning in both policies; a conditional result @@ -538,7 +539,11 @@ worktree or a submodule. Blockers are unmet dependency IDs or one of `work list` fails explicitly if an unfinished formalizable leaf lacks one; plan the missing IDs with `autoform migrate article-ids` and add them to the frontmatter. `work context` may still select that article by its path ID to -report the migration blocker. Both commands are read-only projections of +report the migration blocker. An item whose article records `statement: +retracted` is a revision: it carries `revision` true in JSON, and the text of +`work list` adds a `revision:` line and `work context` a `Revision:` line saying +to start from `autoform work impact` (see the [revision +contract](#revision-contract)). Both commands are read-only projections of Markdown. Under the open policy the text output of `work list` starts with an `Open @@ -558,14 +563,18 @@ autoform work assumptions blueprint --json ``` `work assumptions` prints the policy, one `open:` line per open statement with -its declarations and the open statements it assumes, if any, and one -`conditional:` line per other article whose Lean rests on open statements. +its declarations and the open statements it assumes, if any, one +`conditional:` line per conditional article, and one `unproved:` line per other +listed article that assumes open statements, such as a retracted definition +whose body reaches one. `--json` writes the `autoform-assumptions/v1` contract that CI audits the build -against: every article whose `lean:` names a declaration, stated or not, except -`mathlib: true` ones, with `open`, `assumes`, and `allowed_open_declarations`, -the declarations of the open statements its Lean may reach, plus its own when -it is open. Under the strict policy every such article is listed with `open` -false and nothing allowed. It reads Markdown only and needs no Lean build. +against: every article whose `lean:` names a declaration, stated or not, with +`open`, `assumes`, and `allowed_open_declarations`, the declarations of the +open statements its Lean may reach, plus its own when it is open. A `mathlib: +true` article is listed with state `mathlib`, `open` false, and nothing assumed +or allowed, so CI checks that its names exist and reach no open statement. Under +the strict policy every such article is listed with `open` false and nothing +allowed. It reads Markdown only and needs no Lean build. Ask what revising an article's Lean declarations would affect before editing them: @@ -601,7 +610,10 @@ SECONDS` sets the probe's budget, 600 seconds by default. The report lists: articles whose `lean:` names it, and in `deprecated_unused` those with no users and no such article; - the claim targets: the revised article's first, then every impacted - article's and every helper owner's. + article's, every helper owner's, and, for a helper without an owner, a + `lean/-` key derived from its name, so two revisions touching + the same helper contend for the same claim; each helper reports its key as + `claim_target`. A revision is `contained` when no other article uses it and every helper it impacts belongs to the revised article; it can then be made in place under @@ -610,7 +622,12 @@ it in its statement changes with it, but `work impact` does not report that declaration's users: the additive form `to_additive` writes, which goes unreported itself, or a direction `alias ⟨mp, mpr⟩ :=` takes of an `Iff`, which shows only as proof-impacted. Check such derivations on the revised set by -hand. `--json` writes `autoform-impact/v1`. The [revision +hand. The probe compares only types and values, so it cannot see a change to an +instance's priority or scope, to an attribute such as `@[simp]` or `@[ext]`, or +to notation: such a revision is never contained, whatever `contained` says, and +its claim targets are incomplete, so treat every stated article whose Lean +imports the changed module, directly or not, as statement-impacted. `--json` +writes `autoform-impact/v1`. The [revision contract](#revision-contract) says what to do with the answer. Plan durable article identity metadata without changing the blueprint: @@ -744,21 +761,30 @@ By default a project runs the strict policy: CI rejects every `sorry`, so a theorem's statement lands only together with its proof, and a statement waits until the proof's prerequisites are proved. `open_statements: allowed` in `roadmap/README.md` lets a theorem's statement land with a `sorry` proof. Such -an article, statement formalized and proof not, is an open statement. A proof -that uses open statements is conditional: it compiles and records `proof: -formalized`, but status, `work`, the site, and CI report it as conditionally -proved, never as fully proved. The policy lets dependents be stated and proved -against a faithful statement before its proof exists; the price is conditional +an article, a theorem with `lean:` whose statement is formalized and whose proof +is not, is an open statement. A proof that uses open statements is conditional: +it compiles and records `proof: formalized`, but is reported as conditionally +proved, never as fully proved. Status, `work`, and the site derive that from +the Markdown dependencies and CI from what the Lean uses, so CI can print +`sorry-free` for an article the site shows as conditional, when its Markdown +depends on an open statement its Lean does not use. CI is never the looser of +the two, since a Lean reach the Markdown does not declare fails. The policy lets +dependents be stated and proved against a faithful statement before its proof +exists; the price is conditional results that stay incomplete until every open statement they rest on is proved. -A theorem that loses `statement` while its `lean:` still names a declaration, -as a retraction or a revision leaves it, stays an open statement: that Lean, +A retracted theorem, one recording `statement: retracted` as a revision leaves +it, stays an open statement: the Lean its `lean:` names, `sorry` or not, still compiles into whatever uses it. The audit keeps accepting -its `sorry`, and what rests on it stays conditional, until its proof is -recorded or its `lean:` is removed. A definition is never open: its body +its `sorry`, and what rests on it stays conditional, until Formalize restates +and proves it, or the marker and `lean:` are removed. A theorem that was never +stated is not open even when it has `lean:`: a draft name does not become an +assumption, and CI rejects its `sorry`. A definition is never open: its body is its proof, CI rejects a `sorry` in it, and its statement phase waits until -its proof prerequisites are stated. Turn the policy back off only once no open -statement remains, since the strict audit rejects every `sorry` and strict +its proof prerequisites are stated. A retracted definition is not open either, +but what rests on it still assumes the open statements its body reaches. Turn +the policy back off only once no open statement remains, since the strict audit +rejects every `sorry` and strict status shows a proof resting on one as proved, not conditional. Write an open statement's proof as exactly `sorry`. The audit accepts a `sorry` @@ -797,7 +823,7 @@ of the last three, which read as passing: ```text open statement (proof is sorry): NAME [ID] open statement (proof depends on sorry elsewhere): NAME [ID] -open statement (proof is sorry-free; record proof: formalized): NAME [ID] +open statement (proof is sorry-free; restate it if retracted, then record proof: formalized): NAME [ID] conditional: NAME [ID] rests on open statement(s) A, B sorry-free: NAME [ID] ``` @@ -842,8 +868,10 @@ cleanup. A heartbeat verifies ownership on entry and permanently records any later refusal or transport uncertainty as lost ownership. A claim key is a slug and digest of any string, not a validated node id, so a -shared resource is locked the same way a node is. Parallel agents get one Git -worktree each and serialize `lake build` behind a `lake-build` claim, because +shared resource is locked the same way a node is, as is a Lean helper no +article owns under the `lean/-` key `work impact` reports. +Parallel agents get one Git worktree each and serialize `lake build` behind a +`lake-build` claim, because builds share the elan toolchain and the Mathlib cache even when the checkouts are separate. @@ -876,37 +904,55 @@ Markdown (step 6); Formalize carries out the Lean side (steps 1 to 5). `--declaration` when only some of R's declarations, or a helper, change. 2. Choose the route: - **Contained** (`contained: true`): revise X in place under R's claim. + `contained` ignores R's own declarations, so re-check R's other + declarations that use X as well. - **Expand, migrate, contract**, the default whenever anything uses X: add X' with the revised statement, leave X unchanged and mark it `@[deprecated X' (since := "YYYY-MM-DD")]`, and point R's `lean:` at X'. Under the open policy, while X's proof is still `sorry`, R's `lean:` names X beside X', so the audit keeps accepting that `sorry` as an open statement, and R records `proof` only after step 4 deletes X. - Statement-impacted articles lose `statement` and `proof` but keep `lean:`, - so they return to the frontier; under the open policy a statement-impacted + Statement-impacted articles replace `statement: formalized` with + `statement: retracted`, lose `proof`, and keep `lean:`, so they return to + the frontier as revisions; under the open policy a statement-impacted theorem stays an open statement meanwhile (see [open - statements](#open-statements)). Proof-impacted articles keep everything, - since their proofs still use the valid old X; migrating them to X' is - later work. Claim R and the statement-impacted articles, whose - frontmatter changes. + statements](#open-statements)). When X's proof is sorry-free, + proof-impacted articles keep everything, since their proofs still use the + valid old X; migrating them to X' is later work. While it is still + `sorry`, they lose `proof: formalized` but keep `statement` and `lean:`, + so they return to the frontier as proof phases and migrate to X'; + otherwise they would keep resting on the deprecated, `sorry`'d X, show + "conditional, assumes R" although R's text now describes X', and keep X + out of `deprecated_unused`, so R could never record its proof. The claim + set is every article whose frontmatter changes: R, the statement-impacted + articles, and, in that case, the proof-impacted ones. - **In place**, only when X and X' cannot coexist, for example an instance - or a structure change: claim every `claim_targets` entry and repair every - impacted declaration in one commit whose default build passes. A - statement-impacted article keeps `statement` only after an Agent Review of - its source faithfulness under X's new meaning; otherwise it loses - `statement` and `proof` but keeps `lean:`. A repaired dependent proof - keeps `proof: formalized` only after an Agent Review of the repair; - otherwise it loses `proof`. A theorem's proof that cannot be repaired - becomes exactly `sorry` under the open policy; otherwise delete the - declaration and remove its article's `lean:`, `statement`, and `proof`, - which works only when nothing else uses it. When neither applies, the + or a structure change: the claim set is every `claim_targets` entry. + Repair every impacted declaration in one commit whose default build + passes. A change `work impact` cannot see, to an instance's priority or + scope, an attribute, or notation, also takes this route whatever + `contained` says, and its `claim_targets` are incomplete: add every stated + article whose Lean imports the changed module and treat it as + statement-impacted. A statement-impacted article keeps `statement` only + after an Agent Review of its source faithfulness under X's new meaning; + otherwise it records `statement: retracted`, loses `proof`, and keeps + `lean:`. A repaired dependent proof keeps `proof: formalized` only after + an Agent Review of the repair; otherwise it loses `proof`. A theorem's + proof that cannot be repaired becomes exactly `sorry` under the open + policy; otherwise delete the declaration and remove its article's + `lean:`, `statement`, and `proof`, which works only when nothing else + uses it. When neither applies, the revision is blocked: release the claims and report it. Record what happened under `## Execution notes` of each touched article. -3. Claim the whole set with one `autoform claim acquire`. When it is refused, - release everything and report the held claim as the blocker. After +3. Claim the route's claim set with one `autoform claim acquire`. When it is + refused, release everything and report the held claim as the blocker. After acquiring, re-run `work impact`; if the set grew, release and start over - with the larger set. Under the open policy, reproduce the CI audit as [open - statements](#open-statements) shows before landing. Land one commit, then + with the larger set. After rebasing onto the current shared branch and + rebuilding, re-run it once more; if the set grew, acquire the whole larger + set in one command under the no-hold-and-wait rule of the [claim + contract](#claim-contract) and repair the new targets before landing. Under + the open policy, reproduce the CI audit as [open statements](#open-statements) + shows before landing. Land one commit, then release every claim. 4. Contract: delete a deprecated X once it appears in `deprecated_unused` of `work impact R . --lean-root .`, meaning no declaration uses it and no @@ -918,13 +964,18 @@ Markdown (step 6); Formalize carries out the Lean side (steps 1 to 5). attribute list; point it at the replacement. The check is lexical, so it misses a later `attribute [deprecated] X`, which the deprecated list of `work impact` does see. Under the open policy the finding is expected for a - superseded X that step 2 keeps in R's `lean:` until step 4. -6. When Roadmap revises an article's statement text, it removes that article's - `statement` and `proof` but keeps `lean:`, which `work impact` needs. It - retracts only that article and the dependents whose Markdown text the - revision rewrites; the Lean-side impact decides every other dependent. - Roadmap edits only Markdown: it records the decision, releases its claims, - and leaves the Lean revision to Formalize. + superseded X that step 2 keeps in R's `lean:` until step 4, so `autoform + audit --lean-root` and `autoform doctor --lean-root` fail in that window by + design, while CI, which runs neither, passes. +6. When Roadmap revises an article, its statement text or only its Lean, it + records the decision and retracts the article: it replaces `statement: + formalized` with `statement: retracted`, removes `proof: formalized`, and + keeps `lean:`, which `work impact` needs, so the article returns to the + frontier as a revision; an article without `lean:` just loses `statement` + and `proof`. It retracts only that article and the dependents whose + Markdown text the revision rewrites; the Lean-side impact decides every + other dependent. Roadmap edits only Markdown: it releases its claims and + leaves the Lean revision to Formalize. ## Local runtime doctor diff --git a/autoform_cli/templates/github/autoform_audit.py b/autoform_cli/templates/github/autoform_audit.py index 2e103bcd..895ef2ad 100755 --- a/autoform_cli/templates/github/autoform_audit.py +++ b/autoform_cli/templates/github/autoform_audit.py @@ -676,7 +676,7 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := else if (← Lean.collectAxioms declName).contains ``sorryAx then logInfo m!"open statement (proof depends on sorry elsewhere): {{declName}} [{{article}}]" else if errors.size == reported && !broken then - logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" + logInfo m!"open statement (proof is sorry-free; restate it if retracted, then record proof: formalized): {{declName}} [{{article}}]" else if !hits.isEmpty then conditionalCount := conditionalCount + 1 if errors.size == reported && !broken then diff --git a/skills/formalize/SKILL.md b/skills/formalize/SKILL.md index 3e15fbe4..c2d2181b 100644 --- a/skills/formalize/SKILL.md +++ b/skills/formalize/SKILL.md @@ -63,9 +63,10 @@ phase: do not modify another article or its Lean declarations, or weaken a public statement. The one exception is revising a declaration other articles' Lean uses: start from `autoform work impact` and make only the edits the [revision contract](../../autoform_cli/README.md#revision-contract) requires, -under the claims it requires. A statement phase whose `lean:` already names a -compiled declaration that other articles use, as a Roadmap retraction leaves -it, is such a revision. Search the pinned Mathlib checkout before adding +under the claims it requires. A work item flagged `revision`, whose article +records `statement: retracted`, is such a revision; restating it replaces +`statement: retracted` with `statement: formalized`. Never add a new use of a +deprecated declaration. Search the pinned Mathlib checkout before adding helpers, and use the shared Lean LSP and REPL with `` as the project path. Finish with the focused Lake target. Declare the result in a module the library root imports, or that the lakefile's globs cover, because the default @@ -119,15 +120,20 @@ changing the DAG. Run `autoform check /blueprint --lean-root ` and `autoform audit /blueprint --lean-root `. Resolve every finding this -work introduced on the claimed article; report unrelated pre-existing findings -instead of fixing them. +work introduced on the claimed article, except `lean-target-deprecated` for a +superseded declaration that an expand, migrate, contract revision keeps in the +revised article's `lean:` until it is deleted; report unrelated pre-existing +findings instead of fixing them. Commit the verified result in its worktree, renew the claim, and rebase onto or merge the current shared branch. On the result, run the default `lake build`, confirm that dependency readiness is unchanged, and confirm that the claimed article differs from its starting `article_revision` only by this worker's edits; if integration changed the candidate, repeat the review, check, and -audit. Keep the article claim until every checkout on the claim board can see +audit. For a revision, also re-run `autoform work impact` on the rebuilt result; +if the route's claim set grew, acquire the whole larger set in one command under +the no-hold-and-wait rule and repair the new targets before landing. Keep the +article claim until every checkout on the claim board can see the verified commit: on the shared branch and, for an `origin` board, pushed, since other clones read their frontier from the remote. Without authority to update or push that branch, or when integration fails, keep the claim and report diff --git a/skills/human-review/SKILL.md b/skills/human-review/SKILL.md index 7dd118a7..00c8f105 100644 --- a/skills/human-review/SKILL.md +++ b/skills/human-review/SKILL.md @@ -37,9 +37,9 @@ individual node and Lean-source links. Record each human decision as `approve`, `block`, with the exact page or node and rationale. Separate validator output from the person's judgment. Do not silently apply requested revisions: hand mathematical-plan changes and Lean implementation changes to Roadmap, which -records the decision and leaves the Lean side to Formalize under the [revision -contract](../../autoform_cli/README.md#revision-contract), and autonomous -rubric scoring to Agent Review. +records the decision and retracts the affected article so Formalize takes it up +under the [revision contract](../../autoform_cli/README.md#revision-contract), +and autonomous rubric scoring to Agent Review. Treat the landing page's `Scoped roadmap` percentage as completion among formalizable leaf targets that are fully proved, including every dependency diff --git a/skills/roadmap/SKILL.md b/skills/roadmap/SKILL.md index 59b0d33d..40c8b10d 100644 --- a/skills/roadmap/SKILL.md +++ b/skills/roadmap/SKILL.md @@ -74,14 +74,18 @@ context` reports for it, passing your own `--worker-id`; renew it while editing and release it once the committed revision is on the branch Formalize works from. A refused acquire means another agent owns the article: leave it and report it. Claims write refs to the board's remote, which is outward-facing, so -make sure the request covers them. When a revision changes a statement, remove -its `statement` and `proof` metadata but keep `lean:`, which Formalize needs to -run `autoform work impact`; under the open policy a retracted theorem stays an -open statement, so whatever rests on it stays conditionally proved. Retract only -that article and the dependents whose Markdown text the revision rewrites, +make sure the request covers them. When a revision changes a statement whose +article has `lean:`, retract it: replace `statement: formalized` with +`statement: retracted`, remove `proof: formalized`, and keep `lean:`, which +Formalize needs to run `autoform work impact`; an article without `lean:` just +loses `statement` and `proof`. Under the open policy a retracted theorem stays +an open statement, so whatever rests on it stays conditionally proved. Retract +only that article and the dependents whose Markdown text the revision rewrites, claiming them all in one acquire; the Lean-side impact decides every other dependent. For a Lean revision requested in Human Review, record the decision in -the article and leave the Lean change to Formalize, which follows the [revision +the article and retract it the same way, so it returns to the frontier as a +statement phase flagged as a revision, and leave the Lean change to Formalize, +which follows the [revision contract](../../autoform_cli/README.md#revision-contract); this skill edits only Markdown. For a large source, divide independent sections among available agents while retaining one owner for global coverage and dependency diff --git a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py index 2e103bcd..895ef2ad 100755 --- a/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py +++ b/skills/setup/assets/cabannes-thesis-project/.github/autoform_audit.py @@ -676,7 +676,7 @@ def autoformOpenAuditNameList (names : Array Name) : MessageData := else if (← Lean.collectAxioms declName).contains ``sorryAx then logInfo m!"open statement (proof depends on sorry elsewhere): {{declName}} [{{article}}]" else if errors.size == reported && !broken then - logInfo m!"open statement (proof is sorry-free; record proof: formalized): {{declName}} [{{article}}]" + logInfo m!"open statement (proof is sorry-free; restate it if retracted, then record proof: formalized): {{declName}} [{{article}}]" else if !hits.isEmpty then conditionalCount := conditionalCount + 1 if errors.size == reported && !broken then From cc36d27ec34f7615c617f36fefebaf05c7e033be Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 15:29:26 -0400 Subject: [PATCH 35/38] Claim a revised declaration no article names like an unowned helper `work impact R --declaration X` skipped X when listing helpers, so when no article named X its claim set held only R's target. Two revisions of the same unowned helper from different articles then each claimed only their own article and could both edit X. A revised declaration no article names is now claimed under its owner's target, or under the `lean/-` key an unowned helper gets, and `contained` holds exactly when the revised article's target is the only one, so such a revision is never reported as safe to make in place under R's claim alone. --- autoform_cli/README.md | 6 ++++-- autoform_cli/impact.py | 16 ++++++++++------ tests/test_impact.py | 34 ++++++++++++++++++++++++++++++++-- 3 files changed, 46 insertions(+), 10 deletions(-) diff --git a/autoform_cli/README.md b/autoform_cli/README.md index 3cebdb6d..bfb73922 100644 --- a/autoform_cli/README.md +++ b/autoform_cli/README.md @@ -613,10 +613,12 @@ SECONDS` sets the probe's budget, 600 seconds by default. The report lists: article's, every helper owner's, and, for a helper without an owner, a `lean/-` key derived from its name, so two revisions touching the same helper contend for the same claim; each helper reports its key as - `claim_target`. + `claim_target`. A revised declaration that no article names is claimed the + same way, under its owner's target or its own key. A revision is `contained` when no other article uses it and every helper it -impacts belongs to the revised article; it can then be made in place under +impacts or revises belongs to the revised article, so its only claim target is +that article's; it can then be made in place under that article's claim. A declaration derived from a revised one without naming it in its statement changes with it, but `work impact` does not report that declaration's users: the additive form `to_additive` writes, which goes diff --git a/autoform_cli/impact.py b/autoform_cli/impact.py index 715fa9bb..a749da2a 100644 --- a/autoform_cli/impact.py +++ b/autoform_cli/impact.py @@ -398,14 +398,13 @@ def contained(self) -> bool: A helper the revised article owns, such as a structure's generated constructor or recursor, is repaired under that article's claim, so it - does not count; any other helper, owned or not, does. + does not count; any other helper, owned or not, does, and so does a + revised declaration that belongs to another article or to none. The + revision is contained exactly when its only claim target is the + revised article's. """ - return not ( - self.statement_impacted - or self.proof_impacted - or any(helper.owner != self.article.id for helper in self.helpers) - ) + return self.claim_targets == (self.article.claim_target,) def as_dict(self) -> dict[str, object]: return { @@ -542,7 +541,12 @@ def compute_impact( # A helper is repaired under its owner's claim, so the owner is claimed # too; an unowned helper is repaired under a claim keyed by its own name. + # A revised declaration no article names is claimed the same way. others = {item.claim_target for item in impacted} | {helper.claim_target for helper in helpers} + for name in revised_names: + if name not in named: + owner = _owner(records[name], records, named) + others.add(by_id[owner].claim_target if owner in by_id else _helper_claim_key(name)) claim_targets = (revised.claim_target, *sorted(others - {revised.claim_target})) return ImpactReport( source_revision=source_revision, diff --git a/tests/test_impact.py b/tests/test_impact.py index 41607bcf..b349f6c2 100644 --- a/tests/test_impact.py +++ b/tests/test_impact.py @@ -532,6 +532,35 @@ def test_revisions_touching_one_unowned_helper_contend_for_its_claim() -> None: assert json.loads(right.to_json())["helpers"][0]["claim_target"] == key +def test_a_revised_declaration_no_article_names_is_claimed_like_a_helper() -> None: + records = _records( + _rec("A.R", "def"), + _rec("A.R.aux", "def", parent="A.R"), + _rec("A.S", "def"), + _rec("A.T", "inductive"), + _rec("A.T.aux", "def", parent="A.T"), + _rec("A.loose", "def"), + ) + articles = [_article("r", "A.R"), _article("s", "A.S"), _article("t", "A.T")] + + from_r = _impact(records, articles, "r", ["A.loose"]) + from_s = _impact(records, articles, "s", ["A.loose"]) + owned = _impact(records, articles, "r", ["A.T.aux"]) + own = _impact(records, articles, "r", ["A.R.aux"]) + + # Nothing uses A.loose, yet two revisions of it contend for the claim + # keyed by its name; a revised declaration another article owns is + # claimed under that article, and one the revised article owns adds no + # claim. + assert from_r.claim_targets == ("r", _lean_key("A.loose")) + assert from_s.claim_targets == ("s", _lean_key("A.loose")) + assert not from_r.contained + assert owned.claim_targets == ("r", "t") + assert not owned.contained + assert own.claim_targets == ("r",) + assert own.contained + + def test_an_unowned_helper_claim_key_is_ref_safe_for_any_name() -> None: names = ("_private.Demo.Extra.0.A.priv", "A.«weird name»", "«∀»", "A." + "long" * 20) records = _records(_rec("A.base", "def"), *(_rec(name, type_uses=("A.base",)) for name in names)) @@ -967,8 +996,9 @@ def test_cli_declaration_flag_replaces_the_article_s_names(tmp_path: Path, monke report = json.loads(output.out) assert report["article"] == {"id": "chapter/empty", "article_id": None, "claim_target": "chapter/empty"} assert report["declarations"] == ["Demo.gone"] - assert report["contained"] is True - assert report["claim_targets"] == ["chapter/empty"] + # No article names Demo.gone, so revising it claims its own key too. + assert report["contained"] is False + assert report["claim_targets"] == ["chapter/empty", _lean_key("Demo.gone")] def test_cli_text_escapes_terminal_control_characters(tmp_path: Path, monkeypatch, capsys) -> None: From 433de30312d6d61c4065429eb2b253a7f003f334 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 15:29:38 -0400 Subject: [PATCH 36/38] Close three gaps in the revision and open-statement docs The expand route told proof-impacted articles to drop `proof: formalized` so they return to their proof phase while X is still `sorry`. A stated definition counts as proved whatever `proof:` says, so that did nothing for a proof-impacted definition, which kept resting on the deprecated X and kept it out of `deprecated_unused`. Such a definition is now migrated to X' in the same commit or, failing that, retracted like a statement-impacted article, and the claim set covers every article whose frontmatter or Lean changes. A project whose CI pins an older AUTOFORM_REF fails `autoform check` on `statement: retracted`, since older loaders accept only `formalized`. The README's assertion table and the Roadmap skill now say to move the pin first, and until then to retract by removing `statement` and `proof` and keeping `lean:`. Status treats every stated, unproved theorem as an open statement, with or without `lean:`, while the README defined one as having `lean:`. The README now matches the code and notes that CI audits an open statement only through the declarations its `lean:` names. --- autoform_cli/README.md | 29 +++++++++++++++++------------ skills/roadmap/SKILL.md | 5 ++++- 2 files changed, 21 insertions(+), 13 deletions(-) diff --git a/autoform_cli/README.md b/autoform_cli/README.md index bfb73922..c316452a 100644 --- a/autoform_cli/README.md +++ b/autoform_cli/README.md @@ -84,7 +84,7 @@ An article asserts only facts a human or agent verified: | Key | Meaning | | --- | --- | | `statement: formalized` | The Lean statement exists and compiles. | -| `statement: retracted` | A revision retracted the statement while `lean:` still names the old declaration, which stays in the build until Formalize restates the article and records `statement: formalized` in its place. Requires `lean:`; invalid with `proof: formalized` or `mathlib: true`. | +| `statement: retracted` | A revision retracted the statement while `lean:` still names the old declaration, which stays in the build until Formalize restates the article and records `statement: formalized` in its place. Requires `lean:`; invalid with `proof: formalized` or `mathlib: true`. CI's `autoform check` at an older `AUTOFORM_REF` rejects the marker, so move the pin first; until then, retract by removing `statement` and `proof` and keeping `lean:`. | | `proof: formalized` | The Lean proof compiles. Under the open policy it may rest on open statements, and the article is then conditional; only the derived `fully_proved` means complete and `sorry`-free. | | `mathlib: true` | The result is upstreamed into Mathlib. | | `not_ready: true` | Needs more blueprint work before it can be attempted. | @@ -111,8 +111,8 @@ lands only with its proof; under `open_statements: allowed` it may land with a `proved` and `fully_proved` differ on purpose: a theorem whose own proof compiles but which rests on unfinished work is green, not dark green. `conditional`, labelled "conditionally proved", is the open policy's case of -that: the article is proved, but it reaches an open statement, a theorem whose -`lean:` names a declaration and which is stated or retracted but not proved +that: the article is proved, but it reaches an open statement, a theorem that +is stated or retracted but not proved (see [Open statements](#open-statements)), through its dependencies. An open dependency counts together with whatever its statement prerequisites reach, and a proved dependency passes on @@ -763,8 +763,9 @@ By default a project runs the strict policy: CI rejects every `sorry`, so a theorem's statement lands only together with its proof, and a statement waits until the proof's prerequisites are proved. `open_statements: allowed` in `roadmap/README.md` lets a theorem's statement land with a `sorry` proof. Such -an article, a theorem with `lean:` whose statement is formalized and whose proof -is not, is an open statement. A proof that uses open statements is conditional: +an article, a theorem whose statement is formalized and whose proof is not, is +an open statement; CI audits it only through the declarations its `lean:` +names. A proof that uses open statements is conditional: it compiles and records `proof: formalized`, but is reported as conditionally proved, never as fully proved. Status, `work`, and the site derive that from the Markdown dependencies and CI from what the Lean uses, so CI can print @@ -921,13 +922,17 @@ Markdown (step 6); Formalize carries out the Lean side (steps 1 to 5). statements](#open-statements)). When X's proof is sorry-free, proof-impacted articles keep everything, since their proofs still use the valid old X; migrating them to X' is later work. While it is still - `sorry`, they lose `proof: formalized` but keep `statement` and `lean:`, - so they return to the frontier as proof phases and migrate to X'; - otherwise they would keep resting on the deprecated, `sorry`'d X, show - "conditional, assumes R" although R's text now describes X', and keep X - out of `deprecated_unused`, so R could never record its proof. The claim - set is every article whose frontmatter changes: R, the statement-impacted - articles, and, in that case, the proof-impacted ones. + `sorry`, proof-impacted theorems lose `proof: formalized` but keep + `statement` and `lean:`, so they return to the frontier as proof phases + and migrate to X'. A stated definition counts as proved whatever its + `proof:` says, so a proof-impacted definition is migrated to X' in the + same commit or, when that is not possible, retracted like a + statement-impacted article. Otherwise they would keep resting on the + deprecated, `sorry`'d X, show "conditional, assumes R" although R's text + now describes X', and keep X out of `deprecated_unused`, so R could never + record its proof. The claim set is every article whose frontmatter or + Lean changes: R, the statement-impacted articles, and, in that case, the + proof-impacted ones. - **In place**, only when X and X' cannot coexist, for example an instance or a structure change: the claim set is every `claim_targets` entry. Repair every impacted declaration in one commit whose default build diff --git a/skills/roadmap/SKILL.md b/skills/roadmap/SKILL.md index 40c8b10d..b89cfd72 100644 --- a/skills/roadmap/SKILL.md +++ b/skills/roadmap/SKILL.md @@ -78,7 +78,10 @@ make sure the request covers them. When a revision changes a statement whose article has `lean:`, retract it: replace `statement: formalized` with `statement: retracted`, remove `proof: formalized`, and keep `lean:`, which Formalize needs to run `autoform work impact`; an article without `lean:` just -loses `statement` and `proof`. Under the open policy a retracted theorem stays +loses `statement` and `proof`. When the project's CI pins an `AUTOFORM_REF` +older than the marker, its `autoform check` rejects `statement: retracted`: +remove `statement` and `proof` and keep `lean:` until the pin moves, and report +the old pin. Under the open policy a retracted theorem stays an open statement, so whatever rests on it stays conditionally proved. Retract only that article and the dependents whose Markdown text the revision rewrites, claiming them all in one acquire; the Lean-side impact decides every other From 88e2da466e938f022f47b68b44e1c4ea231cf354 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 15:29:38 -0400 Subject: [PATCH 37/38] Pin the sorry-free open-statement hint and a mathlib contract entry No test checked the audit's line for an open statement whose proof is sorry-free, which now tells the worker to restate a retracted article before recording the proof, and no real-Lean test fed the audit a contract entry in the `mathlib` state the assumptions contract writes. The open-probe fixture gains a sorry-free open theorem whose exact line is asserted, the outside-root declaration is audited as a `mathlib` article, and a `mathlib` article whose Lean reaches an undeclared open statement must fail. --- tests/test_lake_artifact_audit.py | 21 +++++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/tests/test_lake_artifact_audit.py b/tests/test_lake_artifact_audit.py index 6a1e25d4..bdaebb77 100644 --- a/tests/test_lake_artifact_audit.py +++ b/tests/test_lake_artifact_audit.py @@ -483,6 +483,7 @@ def _article( *, is_open: bool = False, allowed: list[str] | None = None, + state: str | None = None, ) -> dict[str, object]: return { "allowed_open_declarations": allowed or [], @@ -491,7 +492,7 @@ def _article( "declarations": declarations, "id": article_id, "open": is_open, - "state": "stated" if is_open else "proved", + "state": state or ("stated" if is_open else "proved"), } @@ -665,6 +666,8 @@ def run(*arguments: str) -> subprocess.CompletedProcess[str]: theorem clean : True := Dep.dep_clean +theorem done_stmt : True := trivial + end Fixture """ @@ -804,7 +807,8 @@ def test_open_probe_accepts_declared_open_statements_and_reductions( _OPEN_ARTICLE, _article("reduction", ["Fixture.reduction"], allowed=["Fixture.open_stmt"]), _article('clean "odd" \\ id', ["Fixture.clean"]), - _article("dependency", ["Dep.dep_clean"]), + _article("mathlib", ["Dep.dep_clean"], state="mathlib"), + _article("done", ["Fixture.done_stmt"], is_open=True, allowed=["Fixture.done_stmt"]), ), ) @@ -813,9 +817,13 @@ def test_open_probe_accepts_declared_open_statements_and_reductions( assert "open statement (proof is sorry): Fixture.open_stmt [open]" in output assert "conditional: Fixture.reduction [reduction] rests on open statement(s) Fixture.open_stmt" in output assert 'sorry-free: Fixture.clean [clean "odd" \\ id]' in output - assert "sorry-free: Dep.dep_clean [dependency]" in output + assert "sorry-free: Dep.dep_clean [mathlib]" in output + assert ( + "open statement (proof is sorry-free; restate it if retracted, then record proof: formalized): " + "Fixture.done_stmt [done]" + ) in output assert "kernel trust clean except declared open statements (" in output - assert "; 1 open statement(s), 1 conditional declaration(s))" in output + assert "; 2 open statement(s), 1 conditional declaration(s))" in output assert "kernel trust clean (" not in output @@ -831,6 +839,11 @@ def test_open_probe_accepts_declared_open_statements_and_reductions( [_article("open", ["Fixture.open_stmt"])], "Fixture.open_stmt contains sorry but is not an open statement", ), + ( + [_OPEN_ARTICLE, _article("mathlib", ["Fixture.reduction"], state="mathlib")], + "Fixture.reduction [mathlib] rests on open statement(s) Fixture.open_stmt, " + "which its article's Markdown dependencies do not reach", + ), ], ) def test_open_probe_rejects_sorry_the_markdown_does_not_declare( From 0237cd6633f59dc21c12bf788c00a0e22abcdf35 Mon Sep 17 00:00:00 2001 From: Jack McCarthy <37917934+Deicyde@users.noreply.github.com> Date: Mon, 5 Oct 2026 14:55:06 -0400 Subject: [PATCH 38/38] Pin the wiki/Lean open-statement contract with a property test Three readers interpret open statements: derive, the work list, and the assumptions contract CI audits Lean against. Each was pinned only by hand-written fixtures, so a change to one could drift from the others without any test noticing. tests/test_contract.py writes seeded random roadmaps of theorems and definitions under both policies and checks them against each other and against the stated semantics: fully proved rests on nothing open, conditional is exactly proved-but-assuming, the work list offers exactly the ready phases, the contract opens exactly the open statements and allows only what each article rests on, and the audit's validator accepts every contract. Invalid retractions are generated too and must fail to load. A real-Lean test builds a small open-policy project, runs the audit the way the verify workflow does, and checks that the site never claims more than the audit: every fully proved article is sorry-free, and every article resting on an open statement is conditional or open. It joins the real-Lean CI step, which took under 7 minutes locally on a heavily loaded machine. --- .github/workflows/tests.yml | 2 +- tests/test_contract.py | 471 ++++++++++++++++++++++++++++++++++++ 2 files changed, 472 insertions(+), 1 deletion(-) create mode 100644 tests/test_contract.py diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 15f4d063..34614fc6 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -64,7 +64,7 @@ jobs: env: AUTOFORM_REQUIRE_REAL_LEAN_TESTS: "1" - name: Run the open-statement audit and impact probe against real Lean - run: timeout --signal=TERM --kill-after=30s 10m uv run pytest -q tests/test_lake_artifact_audit.py tests/test_impact.py + run: timeout --signal=TERM --kill-after=30s 10m uv run pytest -q tests/test_lake_artifact_audit.py tests/test_impact.py tests/test_contract.py env: AUTOFORM_REQUIRE_REAL_LEAN_TESTS: "1" diff --git a/tests/test_contract.py b/tests/test_contract.py new file mode 100644 index 00000000..6f7d590b --- /dev/null +++ b/tests/test_contract.py @@ -0,0 +1,471 @@ +"""The contract between the wiki and the Lean build, checked on random roadmaps. + +PR #115 lets a theorem's statement land with a ``sorry`` proof when +``roadmap/README.md`` sets ``open_statements: allowed``. Three readers must then +agree on what that means: the derived status (``derive``), the work frontier +(``list_ready_work``), and the assumptions contract CI audits the Lean build +against (``assumption_contract``). The property test writes seeded random +roadmaps under both policies and checks each reader against the others and +against the semantics stated here; the real-Lean test checks that the published +site never claims more than the audit of an actual build. +""" + +from __future__ import annotations + +import random +import re +import shutil +import subprocess +import sys +from dataclasses import dataclass, replace +from pathlib import Path + +import pytest + +from autoform_cli.graph import GraphValidationError, load_graph +from autoform_cli.lean import declaration_names +from autoform_cli.runtime import load_runtime_graph +from autoform_cli.status import STATES, derive +from autoform_cli.work import WorkError, assumption_contract, list_ready_work +from tests.test_impact import _lean_toolchain_available +from tests.test_lake_artifact_audit import _TEMPLATE, _load_helper, _run, _write + + +_SEEDS = range(30) +_POLICIES = ("forbidden", "allowed") + + +@dataclass(frozen=True) +class _Spec: + """One generated article: its frontmatter and its typed dependency links.""" + + name: str + declaration: str | None = None + statement: str | None = None + proof: bool = False + mathlib: bool = False + not_ready: bool = False + lean: str | None = None + article_id: str | None = None + statement_dependencies: tuple[str, ...] = () + proof_dependencies: tuple[str, ...] = () + + @property + def dependencies(self) -> tuple[str, ...]: + return self.statement_dependencies + tuple( + name for name in self.proof_dependencies if name not in self.statement_dependencies + ) + + +def _generate(rng: random.Random) -> list[_Spec]: + """Return a random roadmap in dependency order; file names are shuffled.""" + count = rng.randint(2, 8) + names = [f"a{index}" for index in rng.sample(range(count), count)] + specs: list[_Spec] = [] + for index, name in enumerate(names): + declaration = rng.choices(["theorem", "def", None], weights=[6, 3, 1])[0] + mathlib = rng.random() < 0.15 + lean = None + if rng.random() < 0.6: + lean = f"Gen.{name}" if rng.random() < 0.8 else f"Gen.{name}, Gen.{name}_aux" + statements = [None, "formalized"] + (["retracted"] if lean and not mathlib else []) + statement = rng.choices(statements, weights=[35, 45, 20][: len(statements)])[0] + statement_dependencies: list[str] = [] + proof_dependencies: list[str] = [] + for earlier in names[:index]: + roll = rng.random() + if roll < 0.2 or 0.35 <= roll < 0.4: + statement_dependencies.append(earlier) + if 0.2 <= roll < 0.4: + proof_dependencies.append(earlier) + specs.append( + _Spec( + name=name, + declaration=declaration, + statement=statement, + proof=statement != "retracted" and rng.random() < 0.35, + mathlib=mathlib, + not_ready=rng.random() < 0.1, + lean=lean, + article_id=f"af_{index + 1:024x}" if rng.random() < 0.95 else None, + statement_dependencies=tuple(statement_dependencies), + proof_dependencies=tuple(proof_dependencies), + ) + ) + return specs + + +def _write_roadmap(project: Path, specs: list[_Spec], policy: str) -> None: + roadmap = project / "blueprint" / "roadmap" + roadmap.mkdir(parents=True) + (roadmap / "README.md").write_text(f"---\nopen_statements: {policy}\n---\n\n# Roadmap\n", encoding="utf-8") + for spec in specs: + metadata = [ + f"{key}: {value}" + for key, value in ( + ("article_id", spec.article_id), + ("declaration", spec.declaration), + ("statement", spec.statement), + ("proof", "formalized" if spec.proof else None), + ("mathlib", "true" if spec.mathlib else None), + ("not_ready", "true" if spec.not_ready else None), + ("lean", spec.lean), + ) + if value is not None + ] + lines = ["---", *metadata, "---", "", f"# Article {spec.name}", "", "A precise statement."] + for heading, targets in ( + ("Depends on", spec.statement_dependencies), + ("Proof depends on", spec.proof_dependencies), + ): + if targets: + lines.extend(["", f"## {heading}", "", *(f"- [{target}]({target}.md)" for target in targets)]) + (roadmap / f"{spec.name}.md").write_text("\n".join(lines) + "\n", encoding="utf-8") + + +def _open_statements(specs: list[_Spec], open_policy: bool) -> dict[str, frozenset[str]]: + """The open statements each article's proof rests on, from the frontmatter alone. + + An open statement is a theorem that is not proved but is stated or records + ``statement: retracted``; its own proof is the ``sorry``. A stated one need + not name ``lean:``, but only those that do appear in the contract, so only + they can be open there. A dependency that is open contributes itself and + what its statement prerequisites reach, a proved dependency or a definition + naming ``lean:`` passes on everything it reaches, and anything else (an + unstated theorem, a Mathlib article) reaches nothing. + """ + reaches: dict[str, frozenset[str]] = {} + assumes: dict[str, frozenset[str]] = {} + for spec in specs: + definition = spec.declaration == "def" + stated = spec.statement == "formalized" or spec.mathlib + proved = spec.proof or spec.mathlib or (definition and stated) + reached = frozenset().union(*(reaches.get(name, frozenset()) for name in spec.dependencies)) + assumes[spec.name] = reached if open_policy and not spec.mathlib else frozenset() + if not open_policy or spec.mathlib: + continue + if proved or (definition and spec.lean): + reaches[spec.name] = reached + elif not definition and (stated or spec.statement == "retracted"): + reaches[spec.name] = frozenset({spec.name}).union( + *(reaches.get(name, frozenset()) for name in spec.statement_dependencies) + ) + return assumes + + +_RETRACTION_FAULTS = { + "no lean": ( + {"lean": None}, + "{name}: statement: retracted needs the lean: declaration it retracts; without lean:, omit statement", + ), + "proof": ({"proof": True}, "{name}: proof: formalized needs statement: formalized, not retracted"), + "mathlib": ({"mathlib": True}, "{name}: a mathlib: true article cannot record statement: retracted"), +} + + +def _check_roadmap( + project: Path, specs: list[_Spec], policy: str, audit_helper, seen: set[str] +) -> None: + open_policy = policy == "allowed" + by_name = {spec.name: spec for spec in specs} + graph = load_graph(project / "blueprint") + assert graph.open_statements is open_policy + assert audit_helper.blueprint_policy(project / "blueprint") == policy + statuses = derive(graph) + assert set(statuses) == {"roadmap", *by_name} + expected_assumes = _open_statements(specs, open_policy) + + def proved_by_frontmatter(name: str) -> bool: + spec = by_name[name] + return spec.proof or spec.mathlib or (spec.declaration == "def" and spec.statement == "formalized") + + def transitive(name: str) -> set[str]: + found: set[str] = set() + pending = list(by_name[name].dependencies) + while pending: + dependency = pending.pop() + if dependency not in found: + found.add(dependency) + pending.extend(by_name[dependency].dependencies) + return found + + open_articles: set[str] = set() + for spec in specs: + status = statuses[spec.name] + seen.add(f"{policy}:{status.key}") + assert status.proved == proved_by_frontmatter(spec.name) + assert status.fully_proved == ( + status.proved and all(statuses[name].fully_proved for name in spec.dependencies) + ) + # 1. Fully proved means every prerequisite, however far down, is proved in + # its own frontmatter, and nothing is assumed. + if status.fully_proved: + assert all(proved_by_frontmatter(name) for name in transitive(spec.name)), spec.name + assert status.assumes == () + # 2. and 3. Assuming an open statement rules out fully proved, and a + # proved article that assumes one is exactly a conditional one. + if status.assumes: + assert not status.fully_proved + assert (status.key == "conditional") == (status.proved and bool(status.assumes)), spec.name + # 4. The strict policy has no open statements to assume. + if not open_policy: + assert status.assumes == () + assert set(status.assumes) == expected_assumes[spec.name], spec.name + assert list(status.assumes) == sorted(status.assumes) + # 10. A theorem that was never stated is nobody's assumption. + if spec.statement is None: + assert all(spec.name not in other.assumes for other in statuses.values()) + # 6. Waiting on nothing means proved, or ready for the next phase. + ready = status.can_prove if status.stated else status.can_state + assert (status.waiting_on == ()) == (status.proved or ready), spec.name + if ( + open_policy + and spec.lean + and not status.proved + and spec.declaration != "def" + and (status.stated or spec.statement == "retracted") + ): + open_articles.add(spec.name) + + runtime = load_runtime_graph(project) + nodes = {node.id: node for node in runtime.nodes} + for spec in specs: + assert nodes[spec.name].dispatchable == (spec.declaration is not None) + + # 5. The work list offers exactly the ready phases derive reports, on + # formalizable leaves with an article_id that are not marked not_ready; an + # unfinished leaf without an article_id stops the whole list instead. + missing = [ + spec.name + for spec in specs + if spec.declaration is not None + and not spec.mathlib + and not statuses[spec.name].proved + and spec.article_id is None + ] + if missing: + seen.add("work:missing-article-id") + with pytest.raises(WorkError, match=re.escape(", ".join(sorted(missing)) + " (plan IDs")): + list_ready_work(project) + else: + frontier = list_ready_work(project) + assert frontier.open_statements is open_policy + offered = {item.node_id: item for item in frontier.items} + expected = { + spec.name: "proof" if statuses[spec.name].stated else "statement" + for spec in specs + if spec.declaration is not None + and spec.article_id is not None + and not spec.not_ready + and not statuses[spec.name].proved + and (statuses[spec.name].can_prove if statuses[spec.name].stated else statuses[spec.name].can_state) + } + assert {name: item.phase for name, item in offered.items()} == expected + for name, item in offered.items(): + seen.add(f"work:{item.phase}") + status = statuses[name] + assert (item.state, item.assumes, item.blockers) == (status.key, status.assumes, ()) + assert item.claim_target == by_name[name].article_id + assert item.revision == (by_name[name].statement == "retracted") + + # 7. to 9. The contract lists every article naming lean:, opens exactly the + # open statements, and allows each article only what it rests on. + contract = assumption_contract(project) + assert contract.open_statements is open_policy + articles = {article.id: article for article in contract.articles} + assert set(articles) == {spec.name for spec in specs if spec.lean} + assert {name for name, article in articles.items() if article.open} == open_articles + open_declarations = {name for article in articles.values() if article.open for name in article.declarations} + for name, article in articles.items(): + spec = by_name[name] + status = statuses[name] + assert article.declarations == tuple(declaration_names(spec.lean or "")) + assert (article.state, article.assumes) == (status.key, status.assumes) + assert set(article.allowed_open_declarations) <= open_declarations, name + # A stated theorem without lean: is still assumed, but names nothing to allow. + allowed = { + declaration + for assumed in status.assumes + if assumed in articles + for declaration in articles[assumed].declarations + } + if article.open: + allowed |= set(article.declarations) + assert article.allowed_open_declarations == tuple(sorted(allowed)), name + if article.open: + seen.add("contract:retracted-open" if spec.statement == "retracted" else "contract:open") + assert set(article.declarations) <= set(article.allowed_open_declarations) + if spec.mathlib: + seen.add("contract:mathlib") + assert (article.open, article.assumes, article.allowed_open_declarations) == (False, (), ()) + if open_policy: + # The audit's own validator accepts every contract the CLI emits. + path = project / "contract.json" + path.write_text(contract.to_json(), encoding="utf-8") + entries = audit_helper.load_assumption_contract(path) + assert {entry.name for entry in entries if entry.is_open} == open_declarations + + +def test_random_roadmaps_keep_the_wiki_and_lean_contract(repo_root: Path, tmp_path: Path) -> None: + audit_helper = _load_helper(repo_root) + seen: set[str] = set() + for seed in _SEEDS: + rng = random.Random(seed) + specs = _generate(rng) + for policy in _POLICIES: + project = tmp_path / f"{seed}-{policy}" + _write_roadmap(project, specs, policy) + try: + _check_roadmap(project, specs, policy, audit_helper, seen) + except AssertionError as error: + raise AssertionError(f"seed {seed}, policy {policy}: {error}") from error + + # 11. Each invalid retraction is refused at load, with its own message. + fault = rng.choice(sorted(_RETRACTION_FAULTS)) + changes, message = _RETRACTION_FAULTS[fault] + victim = rng.randrange(len(specs)) + fields = {"statement": "retracted", "lean": f"Gen.{specs[victim].name}", "proof": False, "mathlib": False} + broken = replace(specs[victim], **{**fields, **changes}) + project = tmp_path / f"{seed}-invalid" + _write_roadmap(project, [*specs[:victim], broken, *specs[victim + 1 :]], rng.choice(_POLICIES)) + with pytest.raises(GraphValidationError) as raised: + load_graph(project / "blueprint") + assert raised.value.issues == (message.format(name=broken.name),), f"seed {seed}" + seen.add(f"invalid:{fault}") + + # The seeds are fixed, so this only guards against a generator change that + # stops exercising a state or a branch. + expected = {f"forbidden:{state.key}" for state in STATES if state.key != "conditional"} + expected |= {f"allowed:{state.key}" for state in STATES} + expected |= {"work:statement", "work:proof", "work:missing-article-id"} + expected |= {"contract:open", "contract:retracted-open", "contract:mathlib"} + expected |= {f"invalid:{fault}" for fault in _RETRACTION_FAULTS} + assert expected <= seen, sorted(expected - seen) + + +# --------------------------------------------------------------------------- # +# Agreement with the audit of a real Lean build +# --------------------------------------------------------------------------- # + + +_CONTRACT_LEAN = """namespace Fixture + +def base : Nat := 2 + +theorem base_eq : base = 2 := rfl + +theorem open_stmt : 1 + 1 = 2 := sorry + +theorem uses_open : 1 + 1 = 2 ∧ True := ⟨open_stmt, trivial⟩ + +theorem retracted_stmt : 3 + 3 = 6 := sorry + +theorem uses_retracted : 3 + 3 = 6 := retracted_stmt + +theorem clean : 2 + 2 = 4 := rfl + +end Fixture +""" + +_CONTRACT_ROADMAP = ( + ("base", ["declaration: def", "statement: formalized", "lean: Fixture.base"], (), ()), + ( + "base-eq", + ["declaration: theorem", "statement: formalized", "proof: formalized", "lean: Fixture.base_eq"], + ("base",), + (), + ), + ("open", ["declaration: theorem", "statement: formalized", "lean: Fixture.open_stmt"], (), ()), + ( + "uses-open", + ["declaration: theorem", "statement: formalized", "proof: formalized", "lean: Fixture.uses_open"], + (), + ("open",), + ), + ("retracted", ["declaration: theorem", "statement: retracted", "lean: Fixture.retracted_stmt"], (), ()), + ( + "uses-retracted", + ["declaration: theorem", "statement: formalized", "proof: formalized", "lean: Fixture.uses_retracted"], + (), + ("retracted",), + ), + ( + "clean", + ["declaration: theorem", "statement: formalized", "proof: formalized", "lean: Fixture.clean"], + (), + (), + ), +) + +_AUDIT_LINE = re.compile(r"(sorry-free|conditional|open statement \([^)]*\)): (\S+) \[([^\]]+)\]") + + +@pytest.mark.skipif(not _lean_toolchain_available(), reason="needs lake and the fixture's Lean toolchain") +def test_the_site_never_claims_more_than_the_audit_of_a_real_build(repo_root: Path, tmp_path: Path) -> None: + project = tmp_path / "project" + project.mkdir() + shutil.copy(repo_root / "tests/fixtures/skeleton-project/lean-toolchain", project / "lean-toolchain") + _write(project / "lakefile.toml", 'name = "Fixture"\ndefaultTargets = ["Fixture"]\n\n[[lean_lib]]\nname = "Fixture"\n') + _write(project / "Fixture.lean", _CONTRACT_LEAN) + _write(project / "blueprint/roadmap/README.md", "---\nopen_statements: allowed\n---\n\n# Roadmap\n") + for index, (name, metadata, depends, proof_depends) in enumerate(_CONTRACT_ROADMAP, start=1): + lines = ["---", f"article_id: af_{index:024x}", *metadata, "---", "", f"# {name}", ""] + for heading, targets in (("Depends on", depends), ("Proof depends on", proof_depends)): + if targets: + lines.extend([f"## {heading}", "", *(f"- [{target}]({target}.md)" for target in targets), ""]) + _write(project / f"blueprint/roadmap/{name}.md", "\n".join(lines)) + + built = _run(project, "lake", "build") + assert built.returncode == 0, built.stdout + built.stderr + archive = project / "root.tgz" + packed = _run(project, "lake", "pack", str(archive)) + assert packed.returncode == 0, packed.stdout + packed.stderr + + # The verify workflow's steps: read the policy, write the contract, prepare + # the open-statement probe, and run it in the project's Lean environment. + helper = [sys.executable, str(repo_root / _TEMPLATE)] + policy = subprocess.run([*helper, "--policy", "blueprint"], cwd=project, capture_output=True, text=True) + assert (policy.returncode, policy.stdout) == (0, "allowed\n"), policy.stderr + contract = project / "contract.json" + contract.write_text(assumption_contract(project).to_json(), encoding="utf-8") + probe = project / "probe.lean" + prepared = subprocess.run( + [*helper, "--open-statements", str(contract), "Fixture", str(archive), str(probe)], + cwd=project, + capture_output=True, + text=True, + ) + assert prepared.returncode == 0, prepared.stderr + audited = _run(project, "lake", "env", "lean", str(probe)) + output = audited.stdout + audited.stderr + assert audited.returncode == 0, output + assert "kernel trust clean except declared open statements (" in output + + reported: dict[str, set[str]] = {} + for verdict, declaration, article in _AUDIT_LINE.findall(output): + reported.setdefault(article, set()).add(verdict.split(" (")[0]) + assert declaration.startswith("Fixture.") + + runtime = {node.id: node for node in load_runtime_graph(project).nodes} + opened = {article.id for article in assumption_contract(project).articles if article.open} + assert {name: node.status.state for name, node in runtime.items() if name != "roadmap"} == { + "base": "fully_proved", + "base-eq": "fully_proved", + "open": "can_prove", + "uses-open": "conditional", + "retracted": "can_state", + "uses-retracted": "conditional", + "clean": "fully_proved", + } + assert opened == {"open", "retracted"} + # Every article the site shows fully proved, the audit found sorry-free. + for name, node in runtime.items(): + if node.status.fully_proved: + assert reported.get(name) == {"sorry-free"}, (name, output) + # Every article the audit found resting on an open statement, the site + # shows conditional or open, never proved outright. + resting = {name for name, verdicts in reported.items() if verdicts & {"conditional", "open statement"}} + assert resting == {"open", "uses-open", "retracted", "uses-retracted"}, output + for name in resting: + assert runtime[name].status.state == "conditional" or name in opened, name + assert not runtime[name].status.fully_proved