diff --git a/autoform_cli/README.md b/autoform_cli/README.md index 5aba7679..c201ea61 100644 --- a/autoform_cli/README.md +++ b/autoform_cli/README.md @@ -790,6 +790,31 @@ its local context. Point `mkdocs.yml` at `docs_dir: site-src` and enable `md_in_html` plus a `pymdownx.superfences` mermaid fence; see the [repository example](../skills/setup/assets/cabannes-thesis-project/mkdocs.yml). +Publication is staged, synced, validated, and atomically exchanged with the +previous generated site. This fail-closed transaction requires macOS +`renameatx_np` or Linux `renameat2`, plus descriptor-relative traversal, +advisory locking, and directory sync. Autoform exercises no-replace and +cross-directory exchange inside its private workspace before it inspects the +live output, so a network or local filesystem that does not implement those +flags fails without changing the published site. Other platforms, including +Windows, can still use the remaining supported CLI commands but cannot run +`autoform render`; Autoform never falls back to a two-rename replacement with a +missing-site crash window. A legacy +`autoform-publication/v1` output is never deleted automatically. Remove it +explicitly or choose an empty output directory once, then subsequent v2 renders +can replace only the exact checksummed generation they inspected. +The renderer hashes both the blueprint snapshot and the exact Lean-file +generation used for declaration links, then rechecks both under the publication +lock immediately before the atomic rename. That check is the publication +linearization point; later source edits belong to the next render. Render +never indexes a complete generated v1/v2 publication tree or one of its private +workspaces as Lean source. +Repository links are emitted only when those captured blueprint and Lean bytes +match one locally available Git commit. Mutable ref names are recorded as that +commit's full object ID. For dirty, untracked, or otherwise unverifiable inputs, +the render remains local, keeps source notes in the site, records no Git ref, +and reports a warning instead of producing a stale or missing link. + ## Validation `autoform check` rejects cycles, missing targets, escaping paths, @@ -1311,8 +1336,29 @@ project, chapter, nested-scope, local, and full-graph scales. It never reads a `graph.json` or an operational queue. Hidden files are omitted, while symlinks, credentials, logs, provider state, and agent/task state inside the blueprint cause the render to fail rather than silently leak them. Source and output -directories must be disjoint. - -Every render writes `publication.json` with the source-content hash, Git ref, -article and dependency counts, and available views. It contains no timestamp or -absolute path, so identical inputs produce identical output files. +directories must be disjoint. The site root holds the generated `SUMMARY.md`, +`dependencies.md`, `structure.md`, `publication.json`, `assets/`, +`dependencies/`, `javascripts/` and `stylesheets/`; a blueprint root entry that +differs from one of these only in case is refused, because the two would collide +on a case-insensitive filesystem. + +Every render writes `publication.json` with blueprint and Lean-source hashes, +Git ref, article and dependency counts, complete file inventory, and available +views. It contains no timestamp or absolute path, so identical inputs produce +identical output files. Autoform validates and syncs the staged tree before one +atomic filesystem commit, then verifies ownership and syncs both parent +directories. Once the commit begins, Autoform never tries to exchange a recovery +path back into the live destination. If it cannot verify the final state or +sync it, it preserves the private workspace instead of deleting a potentially +unique generation. (On macOS, `fsync` does not flush the drive's cache, so a +power loss can still lose the newest generation.) It reports an exact recovery path only while the bound output +parent is still addressable. Losing that parent path after commit is an uncertain +publication error; the parent is checked again after workspace cleanup before +success is returned. Other post-verification cleanup refusals leave the +published site in place and return the retained workspace as a warning. A process +exit immediately after exchange likewise leaves the complete previous generation +under that workspace while the complete replacement occupies the output path. +The workspace is a hidden `.autoform-publication-*` directory beside the output +(the repository root for `--output site-src`). A render killed by a signal +leaves it behind without a message; later renders do not remove it. It may hold +the only copy of the previous generation, so inspect it before removing it. diff --git a/autoform_cli/__main__.py b/autoform_cli/__main__.py index e2e1cf80..612b8dd8 100644 --- a/autoform_cli/__main__.py +++ b/autoform_cli/__main__.py @@ -25,7 +25,7 @@ from .lean import build_linker, declaration_names, index_failure_message from .project import ProjectCatalogError, ProjectCreateError, create_project, inspect_project, load_release_catalog from .render import PublicationError, render_site -from .runtime import RuntimeProjectionError, load_runtime_graph, resolve_runtime_paths +from .runtime import RuntimeProjectionError, resolve_runtime_paths from .scaffold import ScaffoldError, scaffold_project from .search import SearchError, search_blueprint, statement_preview from .skeleton import ( @@ -317,7 +317,11 @@ def main(argv: Sequence[str] | None = None) -> int: render.add_argument("-o", "--output", default="site-src", help="output directory") render.add_argument("--lean-root", type=Path, help="Lean project to link code from") render.add_argument("--repository-url", help="project URL, e.g. https://github.com/owner/repo") - render.add_argument("--ref", help="commit or branch the code links should pin") + render.add_argument( + "--ref", + help="commit or branch to link; links appear only if its local commit " + "holds the exact rendered inputs", + ) render.add_argument( "--require-declarations", action="store_true", @@ -490,7 +494,6 @@ def _dashboard(args: argparse.Namespace) -> int: def run(scratch: Path) -> None: claims = ClaimBoard(repo, "dashboard-readonly", scratch) state = publication_bound_live_state( - lambda: load_runtime_graph(paths.project_root), claims, blueprint_dir=paths.blueprint_dir, site_dir=site, @@ -1075,6 +1078,8 @@ def _render(args: argparse.Namespace) -> int: print(f"{report.output_dir}: {report.pages} pages, {report.nodes} nodes, {report.linked} code links") for issue in report.unresolved: print(f"warning: declaration not found in the Lean sources: {issue}") + for issue in report.warnings: + print(f"warning: {issue}") if report.unresolved and args.require_declarations: return 1 return 0 diff --git a/autoform_cli/coverage.py b/autoform_cli/coverage.py index a8bddedb..156281ae 100644 --- a/autoform_cli/coverage.py +++ b/autoform_cli/coverage.py @@ -10,19 +10,23 @@ import hashlib import json +import os import re from collections import Counter +from collections.abc import Mapping from dataclasses import asdict, dataclass -from pathlib import Path +from pathlib import Path, PurePosixPath from urllib.parse import unquote, urlsplit from .markdown import ( + EXTERNAL_SCHEMES, INLINE_CODE, Content, PublishedTable, content, link_targets, local_target_issue, + markdown_text_anchors, published_tables, rendered_visible_text, ) @@ -142,6 +146,57 @@ def load_coverage(blueprint_dir: str | Path) -> tuple[CoverageSummary | None, tu ) +def load_coverage_snapshot( + blueprint_dir: str | Path, + files: Mapping[str, bytes], +) -> tuple[CoverageSummary | None, tuple[CoverageIssue, ...]]: + """Validate coverage from one immutable captured blueprint generation.""" + + blueprint = Path(blueprint_dir).expanduser() + if not blueprint.is_absolute(): + blueprint = Path.cwd() / blueprint + normalized: dict[Path, bytes] = {} + for raw_relative, data in files.items(): + relative = PurePosixPath(raw_relative) + if ( + not raw_relative + or relative.is_absolute() + or relative.as_posix() != raw_relative + or ".." in relative.parts + ): + return None, (CoverageIssue(0, "coverage snapshot has an invalid path"),) + normalized[_lexical_path(blueprint.joinpath(*relative.parts))] = data + path = _lexical_path(blueprint / "coverage" / "README.md") + content_bytes = normalized.get(path) + if content_bytes is None: + return None, (CoverageIssue(0, "coverage contract is missing"),) + try: + text = content_bytes.decode("utf-8") + except UnicodeError: + return None, (CoverageIssue(0, "coverage contract cannot be read as UTF-8"),) + + rows, issues = _parse_table(text) + issues.extend( + _validate_evidence( + rows, + blueprint=blueprint, + coverage_path=path, + captured_files=normalized, + ) + ) + if issues: + return None, tuple(issues) + return ( + CoverageSummary( + schema=COVERAGE_SCHEMA, + source_path="coverage/README.md", + source_sha256=hashlib.sha256(content_bytes).hexdigest(), + entries=tuple(rows), + ), + (), + ) + + def _parse_table(text: str) -> tuple[list[CoverageEntry], list[CoverageIssue]]: # Only published Markdown can carry the contract. Commented-out and # code-block tables are masked to blank lines first, which keeps every @@ -465,9 +520,14 @@ def _validate_evidence( *, blueprint: Path, coverage_path: Path, + captured_files: Mapping[Path, bytes] | None = None, ) -> list[CoverageIssue]: issues: list[CoverageIssue] = [] - roadmap = (blueprint / "roadmap").resolve() + roadmap = ( + (blueprint / "roadmap").resolve() + if captured_files is None + else _lexical_path(blueprint / "roadmap") + ) for entry in entries: visible_evidence = _visible_markdown(entry.evidence) if not _has_substance(visible_evidence): @@ -493,13 +553,31 @@ def _validate_evidence( # One good link beside a broken one is a broken claim. broken: list[str] = [] for target in targets: - problem = local_target_issue(coverage_path, target, blueprint, label="coverage") + problem = ( + local_target_issue(coverage_path, target, blueprint, label="coverage") + if captured_files is None + else _captured_local_target_issue( + coverage_path, + target, + blueprint, + captured_files, + label="coverage", + ) + ) if problem is not None: broken.append(problem[1]) if broken: issues.extend(CoverageIssue(entry.line, reason) for reason in broken) continue - if not any(_is_roadmap_article(target, coverage_path=coverage_path, roadmap=roadmap) for target in targets): + if not any( + _is_roadmap_article( + target, + coverage_path=coverage_path, + roadmap=roadmap, + captured_files=captured_files, + ) + for target in targets + ): issues.append( CoverageIssue( entry.line, @@ -523,7 +601,13 @@ def _visible_markdown(value: str) -> str: return rendered_visible_text(INLINE_CODE.sub("", value)) -def _is_roadmap_article(target: str, *, coverage_path: Path, roadmap: Path) -> bool: +def _is_roadmap_article( + target: str, + *, + coverage_path: Path, + roadmap: Path, + captured_files: Mapping[Path, bytes] | None = None, +) -> bool: parsed = urlsplit(target) if parsed.scheme or parsed.netloc: return False @@ -531,13 +615,69 @@ def _is_roadmap_article(target: str, *, coverage_path: Path, roadmap: Path) -> b raw_path = unquote(parsed.path) if not raw_path or "\x00" in raw_path: return False - candidate = (coverage_path.parent / raw_path).resolve() + candidate = ( + (coverage_path.parent / raw_path).resolve() + if captured_files is None + else _lexical_path(coverage_path.parent / raw_path) + ) candidate.relative_to(roadmap) - return candidate.is_file() and candidate.suffix.casefold() == ".md" + return ( + candidate.is_file() if captured_files is None else candidate in captured_files + ) and candidate.suffix.casefold() == ".md" except (OSError, RuntimeError, ValueError): return False +def _captured_local_target_issue( + source_path: Path, + target: str, + boundary: Path, + files: Mapping[Path, bytes], + *, + label: str, +) -> tuple[str, str] | None: + split = urlsplit(target) + scheme = split.scheme.casefold() + if scheme in EXTERNAL_SCHEMES: + return None + if scheme: + return f"unsupported-{label}-link", f"{label} link uses unsupported scheme: {target!r}" + if split.netloc: + return f"unsupported-{label}-link", f"{label} link uses a network location: {target!r}" + raw_path = unquote(split.path) + if "\x00" in raw_path: + return f"malformed-{label}-link", f"{label} link contains an invalid path: {target!r}" + if not raw_path: + candidate = _lexical_path(source_path) + else: + relative = Path(raw_path) + if relative.is_absolute(): + return f"{label}-escapes-blueprint", f"{label} link escapes the blueprint: {target!r}" + candidate = _lexical_path(source_path.parent / relative) + boundary = _lexical_path(boundary) + try: + candidate.relative_to(boundary) + except ValueError: + return f"{label}-escapes-blueprint", f"{label} link escapes the blueprint: {target!r}" + data = files.get(candidate) + if data is None: + return f"{label}-not-found", f"{label} link does not resolve to a file: {target!r}" + if split.fragment and candidate.suffix.casefold() == ".md": + try: + text = data.decode("utf-8") + except UnicodeError: + anchors: set[str] = set() + else: + anchors = markdown_text_anchors(text) + if unquote(split.fragment) not in anchors: + return f"{label}-anchor-not-found", f"{label} link fragment does not resolve: {target!r}" + return None + + +def _lexical_path(path: Path) -> Path: + return Path(os.path.normpath(os.fspath(path))) + + def _cells(line: str) -> tuple[str, ...]: stripped = line.strip() if not stripped.startswith("|") or not stripped.endswith("|"): @@ -575,4 +715,5 @@ def _inline_code(value: str) -> str: "CoverageIssue", "CoverageSummary", "load_coverage", + "load_coverage_snapshot", ] diff --git a/autoform_cli/dashboard.py b/autoform_cli/dashboard.py index 2b464829..891dbf85 100644 --- a/autoform_cli/dashboard.py +++ b/autoform_cli/dashboard.py @@ -4,6 +4,7 @@ import json import threading +from dataclasses import dataclass from functools import partial from http.server import SimpleHTTPRequestHandler, ThreadingHTTPServer from pathlib import Path @@ -11,8 +12,14 @@ from urllib.parse import urlsplit from .claims import ClaimTransportError, author_claim_key -from .graph import GraphValidationError -from .render import LIVE_SCRIPT, PUBLICATION_MANIFEST, publication_source_revision +from .graph import GraphValidationError, Node, load_graph_snapshot +from .render import ( + LIVE_SCRIPT, + PUBLICATION_MANIFEST, + PUBLICATION_MANIFEST_MAX_BYTES, + PUBLICATION_SCHEMA, + capture_publication_source, +) from .runtime import RuntimeGraph, RuntimeNode, RuntimeProjectionError @@ -64,13 +71,22 @@ class ClaimReader(Protocol): def list(self) -> list[dict[str, object]]: ... -def build_live_state(runtime: RuntimeGraph, leases: list[dict[str, object]]) -> dict[str, object]: +@dataclass(frozen=True, slots=True) +class _PublicationLiveGraph: + source_revision: str + nodes: tuple[Node, ...] + + +def build_live_state( + runtime: RuntimeGraph | _PublicationLiveGraph, + leases: list[dict[str, object]], +) -> dict[str, object]: """Project live author claims onto nodes without creating durable state.""" # The claim CLI hashes whatever target it is given, so a path-ID claim on an # article that has an article_id is a separate lease. Show both, each with # the target it was taken on. - claims_by_key: dict[str, tuple[RuntimeNode, str]] = {} + claims_by_key: dict[str, tuple[RuntimeNode | Node, str]] = {} for node in runtime.nodes: article_id = getattr(node, "article_id", None) for target in (node.id, article_id) if article_id else (node.id,): @@ -136,7 +152,6 @@ def load() -> dict[str, object]: def publication_bound_live_state( - runtime_loader: Callable[[], RuntimeGraph], claims: ClaimReader, *, blueprint_dir: str | Path, @@ -144,30 +159,52 @@ def publication_bound_live_state( ) -> Callable[[], dict[str, object]]: """Refuse live badges when the built publication is stale or incomplete.""" - load = live_state_loader(runtime_loader, claims) + lock = threading.Lock() blueprint = Path(blueprint_dir) manifest_path = Path(site_dir) / PUBLICATION_MANIFEST def guarded() -> dict[str, object]: - try: - encoded = manifest_path.read_bytes() - if len(encoded) > 64 * 1024: - raise ValueError("publication manifest is too large") - manifest = json.loads(encoded) - if ( - not isinstance(manifest, dict) - or manifest.get("schema") != "autoform-publication/v1" - or manifest.get("complete") is not True - or manifest.get("source_revision") != publication_source_revision(blueprint) - ): - raise ValueError("built dashboard is stale; rerun render and the MkDocs build") - except (OSError, ValueError) as error: - return { - "schema": LIVE_SCHEMA, - "claims": [], - "error": str(error), - } - return load() + with lock: + try: + encoded = manifest_path.read_bytes() + if len(encoded) > PUBLICATION_MANIFEST_MAX_BYTES: + raise ValueError("publication manifest is too large") + manifest = json.loads(encoded) + snapshot = capture_publication_source(blueprint) + if ( + not isinstance(manifest, dict) + or manifest.get("schema") != PUBLICATION_SCHEMA + or manifest.get("complete") is not True + or manifest.get("source_revision") != snapshot.revision + ): + raise ValueError("built dashboard is stale; rerun render and the MkDocs build") + graph = load_graph_snapshot( + snapshot.root, + { + relative.as_posix(): data + for relative, data in snapshot.files.items() + }, + directories=( + "" if relative.as_posix() == "." else relative.as_posix() + for relative in snapshot.directories + ), + ) + runtime = _PublicationLiveGraph( + snapshot.revision, + tuple(graph.nodes.values()), + ) + return build_live_state(runtime, claims.list()) + except ( + ClaimTransportError, + GraphValidationError, + OSError, + ValueError, + ) as error: + return { + "schema": LIVE_SCHEMA, + "claims": [], + "error": f"{type(error).__name__}: {error}", + } return guarded diff --git a/autoform_cli/graph.py b/autoform_cli/graph.py index 86b03f27..57dc6908 100644 --- a/autoform_cli/graph.py +++ b/autoform_cli/graph.py @@ -10,9 +10,11 @@ from __future__ import annotations import hashlib +import os import re +from collections.abc import Callable, Iterable, Mapping from dataclasses import dataclass -from pathlib import Path +from pathlib import Path, PurePosixPath from urllib.parse import unquote, urlsplit from .lean import declaration_names @@ -149,19 +151,71 @@ def load_graph(blueprint_dir: str | Path) -> Graph: if not blueprint.is_dir(): raise GraphValidationError([f"blueprint directory does not exist: {blueprint}"]) + sources, discovery_issues = _discover_nodes(blueprint) + return _load_graph_sources( + blueprint, + sources, + discovery_issues, + canonicalize=lambda path: path.resolve(), + is_file=lambda path: path.is_file(), + ) + + +def load_graph_snapshot( + blueprint_dir: str | Path, + files: Mapping[str, bytes], + *, + directories: Iterable[str], +) -> Graph: + """Load a graph from one already captured immutable blueprint generation.""" + + blueprint = Path(blueprint_dir).expanduser() + if not blueprint.is_absolute(): + blueprint = Path.cwd() / blueprint + normalized_files = { + _snapshot_relative(relative): content for relative, content in files.items() + } + normalized_directories = { + _snapshot_relative(relative, allow_root=True) for relative in directories + } + sources, discovery_issues = _discover_snapshot_nodes( + blueprint, + normalized_files, + normalized_directories, + ) + available = { + _lexical_path(blueprint.joinpath(*relative.parts)) + for relative in normalized_files + } + return _load_graph_sources( + blueprint, + sources, + discovery_issues, + canonicalize=_lexical_path, + is_file=lambda path: path in available, + ) + + +def _load_graph_sources( + blueprint: Path, + sources: list[_NodeSource], + discovery_issues: list[str], + *, + canonicalize: Callable[[Path], Path], + is_file: Callable[[Path], bool], +) -> Graph: issues: list[str] = [] parsed: list[_ParsedNode] = [] canonical_ids: dict[Path, str] = {} node_ids: dict[str, Path] = {} - sources, discovery_issues = _discover_nodes(blueprint) issues.extend(discovery_issues) article_ids: dict[str, str] = {} source_hashes = {source.id: source.source_sha256 for source in sources} - policy_page = (blueprint / "roadmap" / "README.md").resolve() + policy_page = canonicalize(blueprint / "roadmap" / "README.md") open_statements = False for source in sources: - canonical = source.path.resolve() + canonical = canonicalize(source.path) if canonical in canonical_ids: issues.append(f"{source.id}: duplicates node {canonical_ids[canonical]!r}") continue @@ -199,14 +253,21 @@ def load_graph(blueprint_dir: str | Path) -> Graph: if issues: raise GraphValidationError(issues) - parents = _article_parents(parsed) + parents = _article_parents(parsed, canonicalize=canonicalize) nodes: dict[str, Node] = {} for parsed_node in parsed: def resolve(targets: tuple[str, ...], node: _ParsedNode = parsed_node) -> list[str]: resolved: list[str] = [] for target in targets: - dependency, issue = _resolve_target(node, target, blueprint, canonical_ids) + dependency, issue = _resolve_target( + node, + target, + blueprint, + canonical_ids, + canonicalize=canonicalize, + is_file=is_file, + ) if issue: issues.append(issue) elif dependency == node.id: @@ -257,6 +318,25 @@ def resolve(targets: tuple[str, ...], node: _ParsedNode = parsed_node) -> list[s return Graph(blueprint_dir=blueprint, nodes=nodes, open_statements=open_statements) +def _snapshot_relative(value: str, *, allow_root: bool = False) -> PurePosixPath: + if allow_root and value == "": + return PurePosixPath(".") + path = PurePosixPath(value) + if ( + (not value and not allow_root) + or value == "." + or path.is_absolute() + or path.as_posix() != value + or ".." in path.parts + ): + raise GraphValidationError([f"invalid captured blueprint path: {value!r}"]) + return path + + +def _lexical_path(path: Path) -> Path: + return Path(os.path.normpath(os.fspath(path))) + + def _discover_nodes(blueprint: Path) -> tuple[list[_NodeSource], list[str]]: roadmap_root = blueprint / "roadmap" if not roadmap_root.is_dir(): @@ -297,6 +377,61 @@ def _discover_nodes(blueprint: Path) -> tuple[list[_NodeSource], list[str]]: return sources, issues +def _discover_snapshot_nodes( + blueprint: Path, + files: Mapping[PurePosixPath, bytes], + directories: set[PurePosixPath], +) -> tuple[list[_NodeSource], list[str]]: + roadmap_relative = PurePosixPath("roadmap") + if roadmap_relative not in directories: + return [], [f"roadmap directory does not exist: {blueprint / 'roadmap'}"] + + issues: list[str] = [] + sources: list[_NodeSource] = [] + roadmap_root = blueprint / "roadmap" + roadmap_files = sorted( + relative + for relative in files + if len(relative.parts) > 1 and relative.parts[0] == "roadmap" + ) + for relative in roadmap_files: + if relative.name.casefold() == "readme.md" and relative.name != "README.md": + issues.append( + f"{PurePosixPath(*relative.parts[1:])}: noncanonical README filename; " + "container pages must be named exactly README.md for portable behavior " + "on case-sensitive filesystems" + ) + for relative in roadmap_files: + if relative.suffix != ".md": + continue + content = files[relative] + node_relative = PurePosixPath(*relative.parts[1:]) + try: + text = content.decode("utf-8") + except UnicodeError as error: + issues.append(f"{node_relative}: cannot read roadmap page: {error}") + continue + path = blueprint.joinpath(*relative.parts) + node_id = _article_id(path, roadmap_root) + sources.append( + _NodeSource(node_id, path, text, hashlib.sha256(content).hexdigest()) + ) + + chapters: dict[str, list[PurePosixPath]] = {} + for relative in roadmap_files: + if relative.suffix == ".md" and len(relative.parts) > 2: + chapters.setdefault(relative.parts[1], []).append(relative) + for chapter, articles in sorted(chapters.items()): + readme = PurePosixPath("roadmap", chapter, "README.md") + if readme not in files: + issues.append( + f"{chapter}: chapter directory holds {len(articles)} article(s) but no " + f"README.md, so they attach to the roadmap root instead of a chapter; " + f"add {chapter}/README.md with the chapter's H1 title" + ) + return sources, issues + + def _chapter_issues(roadmap_root: Path) -> list[str]: """Reject a chapter directory that names no chapter. @@ -352,9 +487,13 @@ def _article_id(path: Path, roadmap_root: Path) -> str: return relative.with_suffix("").as_posix() -def _article_parents(parsed: list[_ParsedNode]) -> dict[str, str | None]: +def _article_parents( + parsed: list[_ParsedNode], + *, + canonicalize: Callable[[Path], Path] = lambda path: path.resolve(), +) -> dict[str, str | None]: """Infer strict single-parent containment from nested README articles.""" - by_path = {node.path.resolve(): node.id for node in parsed} + by_path = {canonicalize(node.path): node.id for node in parsed} parents: dict[str, str | None] = {} for node in parsed: candidate = node.path.parent @@ -362,7 +501,7 @@ def _article_parents(parsed: list[_ParsedNode]) -> dict[str, str | None]: candidate = candidate.parent parent: str | None = None while candidate != candidate.parent: - readme = (candidate / "README.md").resolve() + readme = canonicalize(candidate / "README.md") if readme in by_path: parent = by_path[readme] break @@ -532,6 +671,9 @@ def _resolve_target( target: str, blueprint: Path, canonical_ids: dict[Path, str], + *, + canonicalize: Callable[[Path], Path] = lambda path: path.resolve(), + is_file: Callable[[Path], bool] = lambda path: path.is_file(), ) -> tuple[str | None, str | None]: split = urlsplit(target) if split.scheme or split.netloc or split.query: @@ -543,10 +685,10 @@ def _resolve_target( if relative.is_absolute() or relative.suffix != ".md": return None, f"{node.id}: dependency target must be a relative .md file: {target!r}" - resolved = (node.path.parent / relative).resolve() + resolved = canonicalize(node.path.parent / relative) if not _is_within(resolved, blueprint): return None, f"{node.id}: dependency target escapes the blueprint directory: {target!r}" - if not resolved.is_file(): + if not is_file(resolved): return None, f"{node.id}: dependency target does not exist: {target!r}" dependency = canonical_ids.get(resolved) if dependency is None: @@ -740,4 +882,5 @@ def _is_within(path: Path, directory: Path) -> bool: "GraphValidationError", "Node", "load_graph", + "load_graph_snapshot", ] diff --git a/autoform_cli/graph_pages.py b/autoform_cli/graph_pages.py index a5debd82..12dbba97 100644 --- a/autoform_cli/graph_pages.py +++ b/autoform_cli/graph_pages.py @@ -27,6 +27,7 @@ NodeLinks = Callable[[Path], Mapping[str, str]] +PageWriter = Callable[[Path, str], None] def write_graph_pages( @@ -35,6 +36,7 @@ def write_graph_pages( destination: str | Path, *, node_links: NodeLinks, + page_writer: PageWriter | None = None, ) -> tuple[Path, ...]: """Write project, chapter, local, and full graph pages. @@ -42,7 +44,9 @@ def write_graph_pages( each generated page. Keeping that callback in the site renderer avoids duplicating its URL and chapter-anchor policy here. """ - destination = Path(destination).resolve() + destination = Path(destination) + if page_writer is None: + destination = destination.resolve() groups = group_nodes(graph) local_views = focus_views(graph, statuses) project_page = destination / "dependencies.md" @@ -93,6 +97,7 @@ def write_graph_pages( f"{project_item_count} item{'s' if project_item_count != 1 else ''} across " f"{len(groups)} chapter{'s' if len(groups) != 1 else ''}." ), + page_writer=page_writer, ) ) @@ -129,6 +134,7 @@ def write_graph_pages( "Dashed chapter boxes stand for external prerequisites or dependents." ), navigation=navigation, + page_writer=page_writer, ) ) @@ -162,6 +168,7 @@ def write_graph_pages( ("Parent map", _markdown_link(parent_page, scope_page)), ("Full theorem DAG", _markdown_link(full_page, scope_page)), ), + page_writer=page_writer, ) ) complete = full_view(graph, statuses) @@ -177,6 +184,7 @@ def write_graph_pages( "prerequisite to what depends on it; dashed arrows are needed only by proofs." ), navigation=_navigation(("Project map", _markdown_link(project_page, full_page))), + page_writer=page_writer, ) ) @@ -204,6 +212,7 @@ def write_graph_pages( "The highlighted item is the current focus." ), navigation=navigation, + page_writer=page_writer, ) ) @@ -225,6 +234,7 @@ def _write_page( lead: str, navigation: str = "", extra: str = "", + page_writer: PageWriter | None = None, ) -> Path: diagram = mermaid.render_view_diagram(view, links=dict(links), include_classdefs=False) sections = [ @@ -244,8 +254,12 @@ def _write_page( sections.extend([f"{lead} {tip}".rstrip(), "", diagram, ""]) if extra: sections.extend([extra, ""]) - page.parent.mkdir(parents=True, exist_ok=True) - page.write_text("\n".join(sections).rstrip() + "\n", encoding="utf-8") + contents = "\n".join(sections).rstrip() + "\n" + if page_writer is None: + page.parent.mkdir(parents=True, exist_ok=True) + page.write_text(contents, encoding="utf-8") + else: + page_writer(page, contents) return page diff --git a/autoform_cli/lean.py b/autoform_cli/lean.py index 8fd81fb6..0b3c4074 100644 --- a/autoform_cli/lean.py +++ b/autoform_cli/lean.py @@ -320,11 +320,13 @@ def open_project_sources( *, exclude_roots: Iterable[str | Path] = (), limits: TreeCaptureLimits = TreeCaptureLimits(), + opaque_markers: Iterable[OpaqueDirectoryMarker] = (), ) -> BoundProjectSources: """Open a retained Lean source root; the caller must close it. A root that changes while it is bound raises ``TreeChangedError``, which a - retry may clear. + retry may clear. Callers may supply bounded opaque markers for generated + directory formats they own; Lean source policy does not name those formats. """ root_path = directory_binding.lexical_absolute_path(root) @@ -340,7 +342,11 @@ def open_project_sources( exclude_roots, root_identity=tree.identity, ) - tree.selection = _lean_tree_selection(excluded, limits=limits) + tree.selection = _lean_tree_selection( + excluded, + limits=limits, + opaque_markers=tuple(opaque_markers), + ) tree.verify() except BaseException: tree.close() @@ -430,6 +436,7 @@ def _lean_tree_selection( excluded: tuple[PurePosixPath, ...], *, limits: TreeCaptureLimits = TreeCaptureLimits(), + opaque_markers: tuple[OpaqueDirectoryMarker, ...] = (), ) -> TreeSelection: return TreeSelection( include=lambda path, mode: _lean_snapshot_includes( @@ -460,6 +467,7 @@ def _lean_tree_selection( _MANAGED_OUTPUT_MANIFEST_BYTE_LIMIT, _is_managed_output_manifest_bytes, ), + *opaque_markers, ), ) @@ -591,8 +599,8 @@ def in_ignored_root(relative_text: str) -> bool: if _path_is_within_roots(relative, ignored_roots): continue relative_path = Path(relative.as_posix()) - _update_source_digest(digest, relative_path, data) source_files.append((relative_path, data)) + _update_source_digest(digest, relative_path, data) try: text = data.decode("utf-8") except UnicodeError: @@ -1000,32 +1008,168 @@ def _committed_source_paths( path: (prefix / path).as_posix() for path, _data in snapshot.source_files } - tree_oids = _git_tree_oids(root, commit, tuple(repo_paths.values())) - if tree_oids is None: + tree_entries = _git_tree_entries(root, commit, tuple(repo_paths.values())) + if tree_entries is None: return None linkable: set[Path] = set() for path, data in snapshot.source_files: framed = b"blob " + str(len(data)).encode("ascii") + b"\0" + data - if hashlib.new(algorithm, framed).hexdigest() == tree_oids.get(repo_paths[path]): + entry = tree_entries.get(repo_paths[path]) + if ( + entry is not None + and entry[0] in {"100644", "100755"} + and entry[1] == "blob" + and hashlib.new(algorithm, framed).hexdigest() == entry[2] + ): linkable.add(path) return frozenset(linkable) -def _git_tree_oids( +def verify_repository_snapshot( + root: str | Path, + requested_ref: str, + files: Iterable[tuple[Path, bytes]], + *, + required_directories: Iterable[Path] = (), + reasons: list[str] | None = None, +) -> str | None: + """Return the immutable commit containing every captured regular file. + + Git repository selectors, replacement refs, alternate object directories, + and command-scoped configuration are stripped by the shared Git runner. + The caller therefore attests the named repository and ordinary commit + graph, rather than an ambient process override. When *reasons* is given, + the first reason verification failed is appended to it, naming paths + relative to the repository only. + """ + + def refuse(reason: str) -> None: + if reasons is not None: + reasons.append(reason) + + try: + repository_root = Path(root).expanduser().resolve() + except (OSError, RuntimeError, ValueError): + return refuse("the repository root could not be resolved") + commit = _git( + repository_root, + "rev-parse", + "--verify", + "--end-of-options", + f"{requested_ref}^{{commit}}", + ) + algorithm = _git(repository_root, "rev-parse", "--show-object-format") + if ( + commit is None + or algorithm not in {"sha1", "sha256"} + or re.fullmatch(r"[0-9a-f]{40}|[0-9a-f]{64}", commit) is None + ): + return refuse(f"{requested_ref!r} does not name a commit in a local Git repository") + prefix = _git(repository_root, "rev-parse", "--show-prefix") + if prefix: + # Paths below are relative to this root, but trees list paths from the + # top level, so no captured file could ever match. + return refuse( + f"the Lean root is the subdirectory {prefix.rstrip('/')} of its Git " + "repository; links need the repository top level" + ) + + captured: dict[str, bytes] = {} + for path, data in files: + try: + relative = path.relative_to(repository_root) + except ValueError: + return refuse("a captured file or directory is outside the repository") + relative_text = relative.as_posix() + if ( + not relative_text + or relative_text == "." + or "\0" in relative_text + or relative.is_absolute() + or ".." in relative.parts + ): + return refuse("a captured path cannot be named in a Git tree") + previous = captured.setdefault(relative_text, data) + if previous != data: + return refuse(f"two captured files claim {relative_text}") + + directories: set[str] = set() + for path in required_directories: + try: + relative = path.relative_to(repository_root) + except ValueError: + return refuse("a captured file or directory is outside the repository") + relative_text = relative.as_posix() + if ( + not relative_text + or relative_text == "." + or "\0" in relative_text + or relative.is_absolute() + or ".." in relative.parts + ): + return refuse("a captured path cannot be named in a Git tree") + directories.add(relative_text) + + entries = _git_tree_entries( + repository_root, + commit, + tuple(sorted(captured)), + ) + if entries is None: + return refuse("the commit's tree could not be read") + for path, data in captured.items(): + entry = entries.get(path) + if ( + entry is None + or entry[0] not in {"100644", "100755"} + or entry[1] != "blob" + ): + return refuse(f"{path} is not a file in commit {commit[:12]}") + framed = b"blob " + str(len(data)).encode("ascii") + b"\0" + data + if ( + hashlib.new(algorithm, framed).hexdigest() != entry[2] + and _git_clean_blob_id(repository_root, path, data) != entry[2] + ): + return refuse(f"{path} differs from commit {commit[:12]}") + directory_entries = _git_tree_entries( + repository_root, + commit, + tuple(sorted(directories)), + recursive=False, + ) + if directory_entries is None or any( + directory_entries.get(path, (None, None, None))[:2] != ("040000", "tree") + for path in directories + ): + return refuse(f"a required directory is not in commit {commit[:12]}") + return commit + + +def _git_tree_entries( root: Path, commit: str, paths: tuple[str, ...], -) -> dict[str, str] | None: - """Read raw tree object IDs for exact paths without Git's quoting layer.""" + *, + recursive: bool = True, +) -> dict[str, tuple[str, str, str]] | None: + """Read exact tree entry modes, kinds, and IDs without Git quoting.""" - result: dict[str, str] = {} + result: dict[str, tuple[str, str, str]] = {} for start in range(0, len(paths), 128): batch = paths[start : start + 128] if not batch: continue try: completed = subprocess.run( - ["git", "ls-tree", "-rz", "--full-tree", commit, "--", *batch], + [ + "git", + "ls-tree", + "-rz" if recursive else "-z", + "--full-tree", + commit, + "--", + *(f":(literal){path}" for path in batch), + ], cwd=str(root), capture_output=True, env=_git_environment(), @@ -1041,9 +1185,16 @@ def _git_tree_oids( continue header, separator, encoded_path = record.partition(b"\t") fields = header.split() - if not separator or len(fields) != 3 or fields[1] != b"blob": + if not separator or len(fields) != 3: continue - result[os.fsdecode(encoded_path)] = fields[2].decode("ascii") + try: + result[os.fsdecode(encoded_path)] = ( + fields[0].decode("ascii"), + fields[1].decode("ascii"), + fields[2].decode("ascii"), + ) + except UnicodeError: + return None return result @@ -1147,6 +1298,31 @@ def _git(root: str | Path, *arguments: str) -> str | None: return output if result.returncode == 0 and output else None +def _git_clean_blob_id(root: Path, path: str, data: bytes) -> str | None: + """Hash captured bytes as Git would store them at *path*. + + A clean checkout can differ from its blobs byte for byte: an LFS smudge + filter or an eol attribute rewrites files on checkout. ``--path`` applies + the same clean filter and eol conversion ``git status`` uses, to the + captured bytes rather than whatever is on disk now. + """ + + try: + result = subprocess.run( + ["git", "hash-object", "--stdin", f"--path={path}"], + cwd=str(root), + input=data, + capture_output=True, + env=_git_environment(), + timeout=10, + check=False, + ) + except (OSError, subprocess.SubprocessError): + return None + output = result.stdout.decode("ascii", "replace").strip() + return output if result.returncode == 0 and output else None + + def _git_environment() -> dict[str, str]: """Make the explicit working directory the only Git repository selector.""" @@ -1178,4 +1354,5 @@ def _git_environment() -> dict[str, str]: "project_source_revision", "snapshot_project_sources", "strip_lean_comments", + "verify_repository_snapshot", ] diff --git a/autoform_cli/markdown.py b/autoform_cli/markdown.py index a48db156..e5f1860a 100644 --- a/autoform_cli/markdown.py +++ b/autoform_cli/markdown.py @@ -584,6 +584,12 @@ def markdown_anchors(path: Path) -> set[str]: text = path.read_text(encoding="utf-8") except (OSError, UnicodeError): return set() + return markdown_text_anchors(text) + + +def markdown_text_anchors(text: str) -> set[str]: + """Return the anchors for already captured Markdown text.""" + lines = text.splitlines() # MkDocs strips YAML frontmatter before Markdown ever sees it, so those # lines cannot contribute headings. Caching by this exact content observes @@ -800,6 +806,7 @@ def _is_within(path: Path, directory: Path) -> bool: "link_targets", "local_target_issue", "markdown_anchors", + "markdown_text_anchors", "PublishedTable", "markdown_links", "mask_fences_and_comments", diff --git a/autoform_cli/render.py b/autoform_cli/render.py index cb44c7ca..38a1548a 100644 --- a/autoform_cli/render.py +++ b/autoform_cli/render.py @@ -9,23 +9,60 @@ from __future__ import annotations +import ctypes +import errno import hashlib import html import json +import os import re -import shutil -from collections.abc import Iterable +import secrets +import stat +import time +import unicodedata +from collections.abc import Iterable, Mapping from dataclasses import dataclass, field from pathlib import Path +from pathlib import PurePosixPath +from types import MappingProxyType +from typing import Callable from urllib.parse import quote, unquote, urlsplit from . import graph_pages, graph_views, mermaid, status -from .coverage import COVERAGE_DISPOSITIONS, CoverageSummary, load_coverage -from .graph import Graph, Node, load_graph -from .lean import SourceLinker, build_linker, declaration_names, index_failure_message +from ._directory_binding import RetainedDirectory +from ._tree_snapshot import ( + BoundDirectoryTree, + OpaqueDirectoryMarker, + TreeCaptureLimitError, + TreeCaptureLimits, + TreeChangedError, + TreeSelection, + TreeSnapshot, + TreeSnapshotError, + bind_directory_tree, +) +from .coverage import COVERAGE_DISPOSITIONS, CoverageSummary, load_coverage_snapshot +from .graph import Graph, Node, load_graph_snapshot +from .lean import ( + BoundProjectSources, + IndexedSourceSnapshot, + SourceLinker, + build_linker, + declaration_names, + detect_ref, + detect_repository_url, + index_failure_message, + open_project_sources, + verify_repository_snapshot, +) from .markdown import article_parts from .status import is_definition +try: + import fcntl +except ImportError: # pragma: no cover - Windows import compatibility + fcntl = None # type: ignore[assignment] + _HEADING = re.compile(r"^ {0,3}(#{1,6})[ \t]+(.+?)[ \t]*#*[ \t]*$") _FENCE = re.compile(r"^ {0,3}(`{3,}|~{3,})") _MARKDOWN_LINK = re.compile(r"(?[^\]]*)\]\(\s*(?P[^)\s]+)(?:\s+[^)]*)?\)") @@ -45,6 +82,35 @@ #: Transcriptions of the paper being formalised. Vault material, not chapters. SOURCES_DIR = "sources" PUBLICATION_MANIFEST = "publication.json" +PUBLICATION_SCHEMA = "autoform-publication/v2" +PUBLICATION_MANIFEST_MAX_BYTES = 8 * 1024 * 1024 +_PUBLICATION_OPAQUE_SCHEMAS = frozenset( + {"autoform-publication/v1", PUBLICATION_SCHEMA} +) +_PUBLICATION_STAGE_PREFIX = ".autoform-publication-" +_WORKSPACE_MARKER = ".autoform-workspace.json" +_WORKSPACE_MARKER_BYTES = ( + b'{"schema":"autoform-publication-workspace/v1"}\n' +) +_PUBLICATION_MAX_ENTRIES = 100_000 +_PUBLICATION_MAX_DEPTH = 128 +_PUBLICATION_MAX_FILE_BYTES = 256 * 1024 * 1024 +_PUBLICATION_MAX_TOTAL_BYTES = 512 * 1024 * 1024 +_PUBLICATION_MANIFEST_MAX_BYTES = PUBLICATION_MANIFEST_MAX_BYTES +_PUBLICATION_CAPTURE_LIMITS = TreeCaptureLimits( + max_entries=_PUBLICATION_MAX_ENTRIES, + max_depth=_PUBLICATION_MAX_DEPTH, + max_file_bytes=_PUBLICATION_MAX_FILE_BYTES, + max_total_bytes=_PUBLICATION_MAX_TOTAL_BYTES, +) +# The Lean walk lists every name under the Lean root, including untracked trees +# such as venv/ or node_modules/ that hold no Lean source, so it bounds depth and +# captured bytes but not the entry count, as `autoform check` does. +_LEAN_CAPTURE_LIMITS = TreeCaptureLimits( + max_depth=_PUBLICATION_MAX_DEPTH, + max_file_bytes=_PUBLICATION_MAX_FILE_BYTES, + max_total_bytes=_PUBLICATION_MAX_TOTAL_BYTES, +) #: Derived views this command rewrites; stale copies must not leak into the site. _GENERATED_FILES = frozenset( { @@ -70,6 +136,70 @@ } ) + +def _is_generated_path(relative: PurePosixPath | Path) -> bool: + """Generated views occupy only the publication root.""" + + return len(relative.parts) == 1 and relative.name in _GENERATED_FILES + + +#: Root entries the renderer writes into every publication. +_GENERATED_ROOT_ENTRIES = frozenset( + { + "SUMMARY.md", + "dependencies.md", + "structure.md", + PUBLICATION_MANIFEST, + "assets", + "dependencies", + "javascripts", + "stylesheets", + } +) + + +def _generated_root_alias(relative: PurePosixPath | Path) -> str | None: + """The generated root entry an authored path differs from only in case. + + The two would collide on a case-insensitive filesystem, so the authored + entry is refused on every platform rather than overwritten or merged. + """ + + first = relative.parts[0] + for generated in _GENERATED_ROOT_ENTRIES: + if first != generated and first.casefold() == generated.casefold(): + return generated + return None + + +def _publication_snapshot_descends(relative: PurePosixPath) -> bool: + if relative.parts and relative.parts[0].casefold() == ".autoform": + return True + return not ( + any(part.startswith(".") for part in relative.parts) + or {part.casefold() for part in relative.parts}.intersection(_LOCAL_ONLY_NAMES) + ) + + +def _publication_snapshot_includes(relative: PurePosixPath, _mode: int) -> bool: + folded_parts = {part.casefold() for part in relative.parts} + name = relative.name.casefold() + return not ( + any(part.startswith(".") for part in relative.parts) + or folded_parts.intersection(_LOCAL_ONLY_NAMES) + or _is_generated_path(relative) + or name == ".env" + or name.startswith(".env.") + or name.endswith((".key", ".log", ".pem")) + ) + + +_PUBLICATION_SNAPSHOT_SELECTION = TreeSelection( + include=_publication_snapshot_includes, + descend=_publication_snapshot_descends, + limits=_PUBLICATION_CAPTURE_LIMITS, +) + #: How a ``declaration:`` value is announced in the statement box. DECLARATION_LABELS = { "abbrev": "Abbreviation", @@ -225,6 +355,7 @@ class RenderReport: nodes: int = 0 linked: int = 0 unresolved: list[str] = field(default_factory=list) + warnings: list[str] = field(default_factory=list) class PublicationError(ValueError): @@ -235,6 +366,212 @@ def __init__(self, issues: Iterable[str]) -> None: super().__init__("; ".join(self.issues)) +class _PublicationRecoveryError(PublicationError): + """A publication result is uncertain and recovery material must be retained.""" + + +@dataclass(frozen=True, slots=True) +class _DestinationState: + """The exact destination generation a render is allowed to replace.""" + + kind: str + identity: tuple[int, int] | None = None + manifest_sha256: str | None = None + directories: tuple[str, ...] = () + files: tuple[tuple[str, str], ...] = () + source_revision: str | None = None + lean_source_revision: str | None = None + + +@dataclass(frozen=True, slots=True) +class _CleanupInventory: + directories: tuple[tuple[str, tuple[int, ...]], ...] + files: tuple[tuple[str, tuple[int, ...], str], ...] + + +@dataclass(frozen=True, slots=True) +class _CleanupIndex: + directories: dict[str, tuple[int, ...]] + files: dict[str, tuple[tuple[int, ...], str]] + children: dict[str, tuple[str, ...]] + + +@dataclass(frozen=True, slots=True) +class _CapturedBlueprint: + root: Path + files: dict[PurePosixPath, bytes] + directories: frozenset[PurePosixPath] + + @classmethod + def from_snapshot(cls, root: Path, snapshot: TreeSnapshot) -> "_CapturedBlueprint": + return cls( + root, + {PurePosixPath(relative): data for relative, data in snapshot.files}, + frozenset(PurePosixPath(relative or ".") for relative in snapshot.directories), + ) + + def path(self, relative: PurePosixPath | Path | str) -> Path: + parts = PurePosixPath(relative).parts + return self.root.joinpath(*parts) + + def relative(self, path: Path) -> PurePosixPath: + relative = _lexical_path(path).relative_to(self.root) + return PurePosixPath(relative.as_posix()) + + def read_text(self, relative: PurePosixPath | Path | str) -> str: + return self.files[PurePosixPath(relative)].decode("utf-8") + + +@dataclass(frozen=True, slots=True) +class PublicationSourceSnapshot: + """Immutable blueprint bytes and layout consumed by a publication.""" + + root: Path + files: Mapping[PurePosixPath, bytes] + directories: frozenset[PurePosixPath] + revision: str + + +@dataclass(slots=True) +class _PublicationPlanBuilder: + root: Path + files: dict[PurePosixPath, bytes] = field(default_factory=dict) + + def _relative(self, path: PurePosixPath | Path | str) -> PurePosixPath: + return _publication_plan_relative(self.root, path) + + def write_bytes(self, path: PurePosixPath | Path | str, data: bytes) -> None: + self.files[self._relative(path)] = bytes(data) + + def write_text(self, path: PurePosixPath | Path | str, text: str) -> None: + self.write_bytes(path, text.encode("utf-8")) + + def read_text(self, path: PurePosixPath | Path | str) -> str: + return self.files[self._relative(path)].decode("utf-8") + + def is_file(self, path: PurePosixPath | Path | str) -> bool: + return self._relative(path) in self.files + + def inventory(self) -> tuple[tuple[str, ...], tuple[tuple[str, str], ...]]: + return _publication_plan_inventory(self.files) + + def freeze(self) -> "_PublicationFilePlan": + self.inventory() + return _PublicationFilePlan(self.root, MappingProxyType(dict(self.files))) + + +@dataclass(frozen=True, slots=True) +class _PublicationFilePlan: + root: Path + files: Mapping[PurePosixPath, bytes] + + def _relative(self, path: PurePosixPath | Path | str) -> PurePosixPath: + return _publication_plan_relative(self.root, path) + + def read_text(self, path: PurePosixPath | Path | str) -> str: + return self.files[self._relative(path)].decode("utf-8") + + def is_file(self, path: PurePosixPath | Path | str) -> bool: + return self._relative(path) in self.files + + def inventory(self) -> tuple[tuple[str, ...], tuple[tuple[str, str], ...]]: + return _publication_plan_inventory(self.files) + + +def _publication_plan_relative( + root: Path, + path: PurePosixPath | Path | str, +) -> PurePosixPath: + if isinstance(path, Path) and path.is_absolute(): + path = path.relative_to(root) + relative = PurePosixPath(path.as_posix() if isinstance(path, Path) else path) + if not _valid_inventory_path(relative.as_posix()): + raise PublicationError([f"invalid planned publication path: {relative}"]) + return relative + + +def _publication_plan_inventory( + planned_files: Mapping[PurePosixPath, bytes], +) -> tuple[tuple[str, ...], tuple[tuple[str, str], ...]]: + directories: set[PurePosixPath] = set() + files: list[tuple[str, str]] = [] + spellings: dict[tuple[str, ...], PurePosixPath] = {} + budget = _InventoryBudget() + for relative, data in sorted( + planned_files.items(), key=lambda item: item[0].as_posix() + ): + budget.add_entry(depth=len(relative.parts)) + budget.add_file(len(data)) + for parent in relative.parents: + if parent != PurePosixPath("."): + directories.add(parent) + for length in range(1, len(relative.parts) + 1): + prefix = PurePosixPath(*relative.parts[:length]) + key = tuple( + unicodedata.normalize("NFC", part).casefold() + for part in prefix.parts + ) + prior = spellings.setdefault(key, prefix) + if prior != prefix: + raise PublicationError( + [f"planned publication paths collide: {prior} and {prefix}"] + ) + if relative.as_posix() != PUBLICATION_MANIFEST: + files.append((relative.as_posix(), hashlib.sha256(data).hexdigest())) + if directories.intersection(planned_files): + raise PublicationError(["planned publication path is both a file and directory"]) + if len(directories) + len(planned_files) > _PUBLICATION_MAX_ENTRIES: + raise PublicationError( + [f"publication tree exceeds max_entries={_PUBLICATION_MAX_ENTRIES}"] + ) + return ( + tuple(sorted(path.as_posix() for path in directories)), + tuple(files), + ) + + +@dataclass(slots=True) +class _InventoryBudget: + entries: int = 0 + total_bytes: int = 0 + + def add_entry(self, *, depth: int) -> None: + if depth > _PUBLICATION_MAX_DEPTH: + raise PublicationError( + [f"publication tree exceeds max_depth={_PUBLICATION_MAX_DEPTH}"] + ) + self.entries += 1 + if self.entries > _PUBLICATION_MAX_ENTRIES: + raise PublicationError( + [f"publication tree exceeds max_entries={_PUBLICATION_MAX_ENTRIES}"] + ) + + def add_file(self, size: int) -> None: + if size > _PUBLICATION_MAX_FILE_BYTES: + raise PublicationError( + [ + "publication tree exceeds " + f"max_file_bytes={_PUBLICATION_MAX_FILE_BYTES}" + ] + ) + self.total_bytes += size + if self.total_bytes > _PUBLICATION_MAX_TOTAL_BYTES: + raise PublicationError( + [ + "publication tree exceeds " + f"max_total_bytes={_PUBLICATION_MAX_TOTAL_BYTES}" + ] + ) + + +@dataclass(slots=True) +class _PublicationCommitState: + """Whether the filesystem commit may have run and was fully verified.""" + + attempted: bool = False + verified: bool = False + + def render_site( blueprint_dir: str | Path, output_dir: str | Path, @@ -244,283 +581,2846 @@ def render_site( ref: str | None = None, clean: bool = True, ) -> RenderReport: - """Write deterministic, read-only projections of the Markdown blueprint. + """Atomically publish deterministic projections of one source generation.""" - Authored Markdown remains the only graph authority. The output joins three - reader surfaces over it: a book, derived progress, and multiscale dependency - maps. Publication excludes hidden and operational files, rejects symlinks, - and never embeds timestamps or machine-specific paths. - """ - blueprint = Path(blueprint_dir).expanduser().resolve() - requested_destination = Path(output_dir).expanduser() - if requested_destination.is_symlink(): + try: + blueprint = Path(blueprint_dir).expanduser().resolve() + requested_destination = Path(output_dir).expanduser() + except (OSError, RuntimeError, ValueError) as error: + raise PublicationError(["publication paths could not be resolved safely"]) from error + if requested_destination.name in {"", ".", ".."}: + raise PublicationError(["output directory must name one ordinary directory"]) + requested_destination = requested_destination.absolute() + try: + destination = requested_destination.parent.resolve() / requested_destination.name + except (OSError, RuntimeError, ValueError) as error: + raise PublicationError(["output directory could not be resolved safely"]) from error + + _require_publication_platform() + if destination.is_symlink(): raise PublicationError(["refusing symlink output directory"]) - destination = requested_destination.resolve() - if _is_within(destination, blueprint) or _is_within(blueprint, destination): + if _publication_paths_overlap(destination, blueprint): raise PublicationError( ["blueprint and output directories must be disjoint; refusing destructive render"] ) - _validate_publication_tree(blueprint) - graph = load_graph(blueprint) - coverage, coverage_issues = load_coverage(blueprint) - if coverage_issues: + try: + with bind_directory_tree( + blueprint, + selection=_PUBLICATION_SNAPSHOT_SELECTION, + require_descriptor=True, + ) as source_tree: + source_snapshot = source_tree.capture() + _validate_publication_snapshot(source_snapshot) + return _render_bound_site( + blueprint, + destination, + source_tree=source_tree, + source_snapshot=source_snapshot, + lean_root=lean_root, + repository_url=repository_url, + ref=ref, + clean=clean, + ) + except TreeCaptureLimitError as error: + raise PublicationError([f"blueprint {error}"]) from error + except TreeSnapshotError as error: + if str(error) == "safe directory traversal is unavailable on this platform": + raise PublicationError( + ["transactional publication requires safe directory traversal"] + ) from error + if not blueprint.is_dir(): + raise PublicationError( + [f"blueprint directory does not exist: {blueprint}"] + ) from error + if isinstance(error, TreeChangedError): + raise PublicationError( + ["blueprint changed during publication; previous site was preserved"] + ) from error + # A lasting failure (an unreadable entry, an unsupported name) carries a + # path relative to the blueprint; a retry would fail the same way. + raise PublicationError([f"blueprint could not be captured: {error}"]) from error + + +def _render_bound_site( + blueprint: Path, + destination: Path, + *, + source_tree: BoundDirectoryTree, + source_snapshot: TreeSnapshot, + lean_root: str | Path | None, + repository_url: str | None, + ref: str | None, + clean: bool, +) -> RenderReport: + """Render captured inputs beside the destination and commit them once.""" + + try: + output_parent = _open_or_create_output_parent(destination.parent) + except OSError as error: raise PublicationError( [ - f"coverage contract line {issue.line}: {issue.reason}" - if issue.line - else f"coverage contract: {issue.reason}" - for issue in coverage_issues + f"could not create and bind the output parent {destination.parent} " + f"safely: {error.strerror or error}" ] + ) from error + try: + return _render_in_bound_output_parent( + blueprint, + destination, + source_tree=source_tree, + source_snapshot=source_snapshot, + lean_root=lean_root, + repository_url=repository_url, + ref=ref, + clean=clean, + output_parent=output_parent, ) - if coverage is None: - raise PublicationError(["coverage contract could not be loaded"]) + finally: + output_parent.close() + + +def _render_in_bound_output_parent( + blueprint: Path, + destination: Path, + *, + source_tree: BoundDirectoryTree, + source_snapshot: TreeSnapshot, + lean_root: str | Path | None, + repository_url: str | None, + ref: str | None, + clean: bool, + output_parent: RetainedDirectory, +) -> RenderReport: + """Render while retaining the exact output-parent generation.""" + + try: + repo_root = ( + Path(lean_root).expanduser().resolve() + if lean_root is not None + else blueprint.parent + ) + except (OSError, RuntimeError, ValueError) as error: + raise PublicationError(["Lean source root could not be resolved safely"]) from error + captured_blueprint = _CapturedBlueprint.from_snapshot(blueprint, source_snapshot) + source_revision = _source_revision(captured_blueprint) + graph, coverage = _load_publication_contract(captured_blueprint) + source_generation_revision = source_snapshot.generation_revision + workspace: Path | None = None + workspace_identity: tuple[int, int] | None = None + workspace_marker: tuple[tuple[int, ...], str] | None = None + workspace_descriptor: int | None = None + remove_workspace = True + commit_state = _PublicationCommitState() + report: RenderReport | None = None + stage_identity: tuple[int, int] | None = None + stage_cleanup_inventory: _CleanupInventory | None = None + stage_descriptor: int | None = None + expected_destination: _DestinationState | None = None + expected_destination_inventory: _CleanupInventory | None = None + lean_sources: BoundProjectSources | None = None + active_failure: BaseException | None = None + try: + workspace, workspace_identity, workspace_marker = _create_workspace( + destination.parent, + output_parent, + ) + workspace_descriptor = _open_workspace_directory( + output_parent, + workspace.name, + workspace_identity, + ) + _probe_publication_filesystem(workspace, workspace_descriptor, output_parent) + _require_output_parent(output_parent, "before destination inspection") + expected_destination = _inspect_destination_at( + output_parent.descriptor, + destination.name, + destination, + allow_os_metadata=True, + ) + if expected_destination.kind != "absent": + assert expected_destination.identity is not None + expected_destination_inventory = _publication_inventory_at( + output_parent.descriptor, + destination.name, + expected_destination.identity, + ) + _require_destination_inventory( + expected_destination_inventory, + expected_destination, + allow_os_metadata=True, + ) + + lean_exclusions = (destination, workspace) + lean_sources, lean_snapshot = _bind_lean_sources( + repo_root, + exclude_roots=lean_exclusions, + ) + lean_source_revision = lean_snapshot.revision + lean_generation_revision = lean_snapshot.generation_revision + resolved_repository_url = ( + repository_url + if repository_url is not None + else detect_repository_url(repo_root) + ) + requested_ref = ref if ref is not None else detect_ref(repo_root) + repository_coordinates_requested = bool( + resolved_repository_url and requested_ref + ) + link_failures: list[str] = [] + resolved_ref = ( + _verified_repository_ref( + repo_root, + requested_ref, + blueprint=captured_blueprint, + lean_snapshot=lean_snapshot, + reasons=link_failures, + ) + if resolved_repository_url and requested_ref + else None + ) + coordinates_stable = ( + repository_url is not None + or detect_repository_url(repo_root) == resolved_repository_url + ) and (ref is not None or detect_ref(repo_root) == requested_ref) + if not coordinates_stable: + resolved_repository_url = None + resolved_ref = None + link_failures.insert(0, "the repository URL or ref changed while rendering") + repository_links_omitted = bool( + not coordinates_stable + or (repository_coordinates_requested and resolved_ref is None) + ) + try: + linker = build_linker( + repo_root, + repository_url=resolved_repository_url, + ref=resolved_ref, + exclude_roots=lean_exclusions, + source_index=lean_snapshot.index, + detect_missing=False, + ) + except (OSError, ValueError) as error: + issue = ( + index_failure_message(error) + if isinstance(error, OSError) + else "Lean sources could not be indexed" + ) + raise PublicationError([issue]) from error + + initial_files: dict[PurePosixPath, bytes] = {} + if not clean and expected_destination.kind == "owned": + initial_files = _read_owned_publication_files_at( + output_parent.descriptor, + destination.name, + expected_destination, + ) + plan, report = _build_publication_plan( + captured_blueprint, + graph, + coverage, + repo_root=repo_root, + linker=linker, + source_revision=source_revision, + lean_source_revision=lean_source_revision, + initial_files=initial_files, + ) + if repository_links_omitted: + report.warnings.append( + "repository links were omitted because the captured blueprint " + "and Lean inputs do not match one verified local Git commit" + + (f": {link_failures[0]}" if link_failures else "") + ) + stage = workspace / "site" + stage_descriptor, stage_identity = _create_stage_directory( + workspace_descriptor, + ) + _materialize_publication_plan(stage_descriptor, plan) + stage_cleanup_inventory = _cleanup_inventory_descriptor(stage_descriptor) + try: + _sync_tree_descriptor(stage_descriptor) + except (OSError, PublicationError) as error: + raise PublicationError( + [ + "publication stage integrity failed before commit; " + "previous site was preserved" + ] + ) from error + + synced_stage_inventory = _cleanup_inventory_descriptor(stage_descriptor) + os.close(stage_descriptor) + stage_descriptor = None + staged = _inspect_destination_at(workspace_descriptor, stage.name, stage) + if ( + staged.kind != "owned" + or staged.identity != stage_identity + or staged.source_revision != source_revision + or staged.lean_source_revision != lean_source_revision + or synced_stage_inventory != stage_cleanup_inventory + ): + raise _PublicationRecoveryError( + [ + "publication stage changed; workspace retained " + f"{_workspace_recovery_location(workspace, output_parent)}" + ] + ) + _require_destination_inventory(stage_cleanup_inventory, staged) + + def require_current_inputs() -> None: + try: + current_source = source_tree.capture() + except TreeSnapshotError as error: + raise PublicationError( + ["blueprint changed during publication; previous site was preserved"] + ) from error + if current_source.generation_revision != source_generation_revision: + raise PublicationError( + ["blueprint changed during publication; previous site was preserved"] + ) + _require_bound_lean_source_revision( + lean_sources, + lean_generation_revision, + ) + + _publish_staged_site( + stage, + destination, + expected_destination, + staged, + expected_inventory=expected_destination_inventory, + staged_inventory=stage_cleanup_inventory, + commit_state=commit_state, + input_guard=require_current_inputs, + output_parent=output_parent, + workspace_identity=workspace_identity, + workspace_descriptor=workspace_descriptor, + ) + report.output_dir = destination + return report + except _PublicationRecoveryError: + remove_workspace = False + raise + except BaseException as error: + active_failure = error + raise + finally: + try: + if commit_state.attempted and not commit_state.verified: + remove_workspace = False + if ( + remove_workspace + and not commit_state.attempted + and stage_identity is not None + and stage_cleanup_inventory is None + and stage_descriptor is not None + ): + try: + stage_cleanup_inventory = _cleanup_inventory_descriptor( + stage_descriptor + ) + except (OSError, PublicationError): + # A changed or unreadable partial stage is not safe to + # delete. The ordinary write/open failures that created a + # stable partial tree are inventoried and removed below. + pass + expected_children: dict[ + str, + dict[tuple[int, int], _CleanupInventory], + ] = {} + if stage_identity is not None and stage_cleanup_inventory is not None: + if commit_state.verified: + if ( + expected_destination is not None + and expected_destination.kind != "absent" + and expected_destination.identity is not None + and expected_destination_inventory is not None + ): + expected_children["site"] = { + expected_destination.identity: expected_destination_inventory, + } + else: + expected_children["site"] = { + stage_identity: stage_cleanup_inventory, + } + parent_changed = not _output_parent_is_current(output_parent) + cleaned = False + if ( + remove_workspace + and not parent_changed + and workspace is not None + and workspace_identity is not None + ): + cleaned = _remove_owned_workspace( + workspace, + workspace_identity, + expected_children=expected_children, + expected_files=( + {_WORKSPACE_MARKER: workspace_marker} + if workspace_marker is not None + else {} + ), + parent_binding=output_parent, + ) + # Cleanup itself performs filesystem operations through the + # retained parent. Its pathname can still be exchanged while that + # work runs, so a pre-cleanup sample cannot justify success. + parent_changed = parent_changed or not _output_parent_is_current( + output_parent + ) + if ( + remove_workspace + and workspace is not None + and parent_changed + and commit_state.verified + ): + recovery_state = ( + "workspace cleanup completed in the original output-parent " + "generation" + if cleaned + else f"workspace {workspace.name} remains in the original " + "output-parent generation" + ) + raise _PublicationRecoveryError( + [ + "publication commit was verified in its bound output parent, " + "but that parent path changed afterward; output location is " + f"uncertain and {recovery_state}" + ] + ) + if remove_workspace and workspace is not None and not cleaned: + issue = ( + ( + "output parent changed; publication workspace " + f"{workspace.name} was retained in the original " + "output-parent generation created at " + f"{workspace.parent}" + ) + if parent_changed + else ( + "publication staging workspace changed; cleanup was refused at " + f"{workspace}" + ) + ) + if commit_state.verified and report is not None: + report.warnings.append(issue) + else: + # An unverified commit only reaches cleanup through a + # failure; str(KeyboardInterrupt()) is empty. + cause = str(active_failure) or type(active_failure).__name__ + raise PublicationError( + [f"publication failed: {cause}; {issue}; workspace was retained"] + ) from active_failure + finally: + if stage_descriptor is not None: + try: + os.close(stage_descriptor) + except OSError: + pass + if workspace_descriptor is not None: + try: + os.close(workspace_descriptor) + except OSError: + pass + if lean_sources is not None: + lean_sources.close() + + +def _build_publication_plan( + blueprint: _CapturedBlueprint, + graph: Graph, + coverage: CoverageSummary, + *, + repo_root: Path, + linker: SourceLinker, + source_revision: str, + lean_source_revision: str, + initial_files: dict[PurePosixPath, bytes], +) -> tuple[_PublicationFilePlan, RenderReport]: + """Build one bounded immutable publication without touching source paths. + + Authored Markdown remains the only graph authority. The output joins three + reader surfaces over it: a book, derived progress, and multiscale dependency + maps. Publication excludes hidden and operational files, rejects symlinks, + and never embeds timestamps or machine-specific paths. + """ + destination = Path("/__autoform_publication_plan__") + plan = _PublicationPlanBuilder(destination, dict(initial_files)) statuses = status.derive(graph) # The repository root, not the vault's parent. A blueprint nested at # /docs/blueprint would otherwise be described as /blueprint, # and every generated permalink would 404. - repo_root = Path(lean_root).expanduser().resolve() if lean_root is not None else blueprint.parent - try: - linker = build_linker(repo_root, repository_url=repository_url, ref=ref) - except OSError as error: - raise PublicationError([index_failure_message(error)]) from error numbers = _number_nodes(graph) used_by = _reverse_edges(graph) - sources_base = _sources_base(blueprint, repo_root, linker) + sources_base = ( + _sources_base(blueprint.root, repo_root, linker) + if PurePosixPath(SOURCES_DIR) in blueprint.directories + else None + ) + + report = RenderReport(output_dir=destination) + node_paths = {_lexical_path(node.path): node for node in graph.nodes.values()} + # Nodes are published as environments on their milestone page, the way a + # blueprint chapter carries many statements in sequence. Each keeps an + # anchor so every cross-reference still lands on the statement itself. + groups = _group_nodes(graph) + containers = _containers(graph) + anchors = { + node_id: _anchor(node_id, group) + for group, node_ids in groups.items() + for node_id in node_ids + } + group_pages = {group: destination / _group_page(group) for group in groups} + targets = { + node_id: (group_pages[group], anchors[node_id]) + for group, node_ids in groups.items() + for node_id in node_ids + } + targets.update( + { + node_id: (destination / node.path.relative_to(blueprint.root), "") + for node_id, node in graph.nodes.items() + if node_id in containers or not node.formalizable + } + ) + node_sources = {_lexical_path(node.path): node_id for node_id, node in graph.nodes.items()} + + for captured_relative, data in sorted( + blueprint.files.items(), key=lambda item: item[0].as_posix() + ): + relative = Path(captured_relative.as_posix()) + if _SKIPPED_DIRECTORIES.intersection(relative.parts) or _is_hidden(relative): + continue + if _is_generated_path(relative): + continue + # Source notes leave the site entirely once readers can reach them in + # the repository, so the book has one reference surface rather than two. + if sources_base is not None and relative.parts[:1] == (SOURCES_DIR,): + continue + source_path = blueprint.path(captured_relative) + target = destination / relative + # Narrative articles remain book pages. Only formalizable leaves are + # consolidated into their containing article with stable anchors. + article = node_paths.get(_lexical_path(source_path)) + if article is not None and article.formalizable and article.id not in containers: + continue + if relative.suffix.lower() == ".md": + rewritten = _rewrite_links( + data.decode("utf-8"), + source_dir=source_path.parent, + page=target, + blueprint=blueprint.root, + destination=destination, + node_sources=node_sources, + targets=targets, + sources_base=sources_base, + ) + plan.write_text(target, rewritten) + else: + plan.write_bytes(target, data) + report.pages += 1 + + overview = destination / "README.md" + if plan.is_file(overview): + plan.write_text( + overview, + _render_landing_page( + plan.read_text(overview), + graph=graph, + statuses=statuses, + coverage=coverage, + groups=groups, + group_pages=group_pages, + page=overview, + destination=destination, + read_page=plan.read_text, + ), + ) + + for group, node_ids in groups.items(): + page = group_pages[group] + narrative = plan.read_text(page) if plan.is_file(page) else None + chapter, linked, unresolved = _render_chapter( + group, + node_ids, + graph=graph, + statuses=statuses, + numbers=numbers, + used_by=used_by, + linker=linker, + page=page, + targets=targets, + narrative=narrative, + blueprint=blueprint.root, + repo_root=repo_root, + destination=destination, + node_sources=node_sources, + containers=containers, + sources_base=sources_base, + source_blueprint=blueprint.root, + read_source=lambda path: blueprint.read_text(blueprint.relative(path)), + ) + plan.write_text(page, chapter) + if narrative is None: # a milestone with no narrative page of its own + report.pages += 1 + report.nodes += len(node_ids) + report.linked += linked + report.unresolved.extend(unresolved) + + book_pages = _book_page_order( + blueprint, + plan, + graph, + ) + # The landing page is a dashboard, not chapter one. Previous/next belongs + # to the book, so the strip starts at the contents page. + _append_book_navigation([p for p in book_pages if p != overview], plan=plan) + structure = destination / STRUCTURE_PAGE + plan.write_text( + structure, + _render_structure_page( + blueprint, + graph, + statuses, + page=structure, + targets=targets, + sources_base=sources_base, + ), + ) + report.pages += 1 + plan.write_text( + destination / "SUMMARY.md", + _render_summary_nav( + book_pages, + destination=destination, + overview=overview, + plan=plan, + ), + ) + + generated_graph_pages = graph_pages.write_graph_pages( + graph, + statuses, + destination, + node_links=lambda page: _anchored_links(targets, page), + page_writer=plan.write_text, + ) + report.pages += len(generated_graph_pages) + + for relative, contents in ( + (STYLESHEET, _stylesheet()), + (MERMAID_SCRIPT, _mermaid_script()), + (LIVE_SCRIPT, _static_asset("blueprint-live.js")), + (LOGO, _logo()), + ): + plan.write_text(destination / relative, contents) + _write_publication_manifest( + plan, + graph, + linker, + coverage=coverage, + complete=True, + source_revision=source_revision, + lean_source_revision=lean_source_revision, + ) + return plan.freeze(), report + + +def _is_os_metadata(relative: str) -> bool: + """Finder writes ``.DS_Store`` into any folder a user browses. + + A publication never contains one, so it cannot be mistaken for authored + output. A live generation may carry it; it leaves with that generation. + """ + + return PurePosixPath(relative).name == ".DS_Store" + + +def _inspect_destination_at( + parent_descriptor: int, + name: str, + display_path: Path, + *, + allow_os_metadata: bool = False, +) -> _DestinationState: + try: + metadata = os.stat(name, dir_fd=parent_descriptor, follow_symlinks=False) + except FileNotFoundError: + return _DestinationState("absent") + except OSError as error: + raise PublicationError(["could not inspect the output directory safely"]) from error + if stat.S_ISLNK(metadata.st_mode): + raise PublicationError(["refusing symlink output directory"]) + if not stat.S_ISDIR(metadata.st_mode): + raise PublicationError(["output path exists and is not a directory"]) + identity = metadata.st_dev, metadata.st_ino + try: + descriptor = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=parent_descriptor, + ) + except OSError as error: + raise PublicationError(["could not inspect the output directory safely"]) from error + try: + if _descriptor_identity(descriptor) != identity: + raise PublicationError(["output directory changed while it was inspected"]) + if _directory_names_match(descriptor, ()): + current = os.stat(name, dir_fd=parent_descriptor, follow_symlinks=False) + if _descriptor_identity(descriptor) != identity or ( + current.st_dev, + current.st_ino, + ) != identity or not _directory_names_match(descriptor, ()): + raise PublicationError(["output directory changed while it was inspected"]) + return _DestinationState("empty", identity=identity) + + try: + manifest_metadata = os.stat( + PUBLICATION_MANIFEST, + dir_fd=descriptor, + follow_symlinks=False, + ) + except FileNotFoundError: + publication = None + manifest_bytes = b"" + else: + if not stat.S_ISREG(manifest_metadata.st_mode): + publication = None + manifest_bytes = b"" + else: + manifest_bytes = _read_regular_file_at( + descriptor, + PUBLICATION_MANIFEST, + display_path / PUBLICATION_MANIFEST, + max_bytes=_PUBLICATION_MANIFEST_MAX_BYTES, + ) + try: + publication = json.loads(manifest_bytes.decode("utf-8")) + except (UnicodeError, json.JSONDecodeError): + publication = None + if not isinstance(publication, dict) or publication.get("schema") != PUBLICATION_SCHEMA: + raise PublicationError( + [ + "refusing to overwrite a non-Autoform output directory or legacy " + "publication; choose an empty directory or remove it explicitly" + ] + ) + canonical_manifest = (json.dumps(publication, indent=2, sort_keys=True) + "\n").encode( + "utf-8" + ) + if manifest_bytes != canonical_manifest: + raise PublicationError(["publication manifest is not in canonical form"]) + if publication.get("complete") is not True: + raise PublicationError(["refusing to overwrite an incomplete Autoform publication"]) + expected_files = _parse_inventory_files(publication.get("files")) + expected_directories = _parse_inventory_directories(publication.get("directories")) + source_revision = publication.get("source_revision") + lean_source_revision = publication.get("lean_source_revision") + if ( + not isinstance(source_revision, str) + or re.fullmatch(r"[0-9a-f]{64}", source_revision) is None + or not isinstance(lean_source_revision, str) + or re.fullmatch(r"[0-9a-f]{64}", lean_source_revision) is None + ): + raise PublicationError(["publication manifest has invalid source revisions"]) + actual_directories, actual_files = _publication_inventory_descriptor(descriptor) + owned_files = tuple( + item + for item in actual_files + if not (allow_os_metadata and _is_os_metadata(item[0])) + ) + if actual_directories != expected_directories: + raise PublicationError( + ["refusing to overwrite an output directory with untracked or missing directories"] + ) + if tuple(path for path, _ in owned_files) != tuple(path for path, _ in expected_files): + expected_paths = {path for path, _ in expected_files} + actual_paths = {path for path, _ in owned_files} + difference = sorted(expected_paths ^ actual_paths) + raise PublicationError( + [ + "refusing to overwrite an output directory with untracked or missing files: " + + ", ".join(difference) + ] + ) + if owned_files != expected_files: + raise PublicationError(["refusing to overwrite a modified Autoform publication"]) + if ( + _read_regular_file_at( + descriptor, + PUBLICATION_MANIFEST, + display_path / PUBLICATION_MANIFEST, + max_bytes=_PUBLICATION_MANIFEST_MAX_BYTES, + ) + != manifest_bytes + ): + raise PublicationError(["publication output changed while it was inspected"]) + current = os.stat(name, dir_fd=parent_descriptor, follow_symlinks=False) + if _descriptor_identity(descriptor) != identity or ( + current.st_dev, + current.st_ino, + ) != identity: + raise PublicationError(["output directory changed while it was inspected"]) + return _DestinationState( + "owned", + identity=identity, + manifest_sha256=hashlib.sha256(manifest_bytes).hexdigest(), + directories=expected_directories, + files=expected_files, + source_revision=source_revision, + lean_source_revision=lean_source_revision, + ) + finally: + os.close(descriptor) + + +def _descriptor_identity(descriptor: int) -> tuple[int, int]: + metadata = os.fstat(descriptor) + if not stat.S_ISDIR(metadata.st_mode): + raise PublicationError(["output path changed while it was inspected"]) + return metadata.st_dev, metadata.st_ino + + +def _stat_signature(metadata: os.stat_result) -> tuple[int, ...]: + # No ctime: a sync or backup agent writing an extended attribute, or our own + # cleanup rename, changes it without changing content. The comparisons that + # guard content also compare a digest. + return ( + metadata.st_dev, + metadata.st_ino, + metadata.st_mode, + metadata.st_nlink, + metadata.st_size, + metadata.st_mtime_ns, + ) + + +def _directory_path_identity(path: Path) -> tuple[int, int]: + descriptor = _open_directory_path(path) + try: + return _descriptor_identity(descriptor) + finally: + os.close(descriptor) + + +def _output_parent_is_current(binding: RetainedDirectory) -> bool: + try: + binding.verify() + except OSError: + return False + return True + + +def _require_output_parent(binding: RetainedDirectory, phase: str) -> None: + if not _output_parent_is_current(binding): + raise PublicationError([f"output parent changed {phase}"]) + + +def _workspace_recovery_location( + workspace: Path, + output_parent: RetainedDirectory, +) -> str: + if _output_parent_is_current(output_parent): + return f"at {workspace}" + return ( + f"as {workspace.name} in the original output-parent generation " + f"created at {workspace.parent}" + ) + + +def _require_publication_platform() -> None: + required_options = ("O_CLOEXEC", "O_DIRECTORY", "O_NOFOLLOW", "O_NONBLOCK") + if ( + fcntl is None + or any(not hasattr(os, option) for option in required_options) + or os.mkdir not in os.supports_dir_fd + or os.open not in os.supports_dir_fd + or os.rename not in os.supports_dir_fd + or os.stat not in os.supports_dir_fd + or os.scandir not in os.supports_fd + ): + raise PublicationError( + ["transactional publication is unavailable on this platform"] + ) + _rename_implementation(exchange=False) + _rename_implementation(exchange=True) + + +def _open_directory_path(path: Path) -> int: + """Open every component without following a symbolic link.""" + absolute = path.absolute() + flags = os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW + descriptor = os.open(absolute.anchor, flags) + try: + for part in absolute.parts[1:]: + child = os.open(part, flags, dir_fd=descriptor) + os.close(descriptor) + descriptor = child + except BaseException: + os.close(descriptor) + raise + return descriptor + + +_SEARCH_ONLY_FLAG = getattr(os, "O_SEARCH", 0) or getattr(os, "O_PATH", 0) + + +def _open_or_create_output_parent(path: Path) -> RetainedDirectory: + """Retain an output-parent chain and durably create missing components.""" + + absolute = path.absolute() + anchor = absolute.anchor + if not anchor: + raise OSError("output parent is not absolute") + flags = os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW + descriptors: list[int] = [] + identities: list[tuple[int, int]] = [] + first_created_index: int | None = None + binding: RetainedDirectory | None = None + try: + descriptor = os.open(anchor, flags) + descriptors.append(descriptor) + opened = os.fstat(descriptor) + named = os.stat(anchor, follow_symlinks=False) + if ( + not stat.S_ISDIR(opened.st_mode) + or not stat.S_ISDIR(named.st_mode) + or (opened.st_dev, opened.st_ino) != (named.st_dev, named.st_ino) + ): + raise OSError(errno.ESTALE, "output-parent anchor changed") + identities.append((opened.st_dev, opened.st_ino)) + + parts = absolute.parts[1:] + for index, part in enumerate(parts): + parent_descriptor = descriptor + created = False + try: + expected = os.stat( + part, + dir_fd=parent_descriptor, + follow_symlinks=False, + ) + except FileNotFoundError: + try: + os.mkdir(part, mode=0o777, dir_fd=parent_descriptor) + except FileExistsError: + pass + else: + created = True + expected = os.stat( + part, + dir_fd=parent_descriptor, + follow_symlinks=False, + ) + if not stat.S_ISDIR(expected.st_mode): + raise OSError(errno.ENOTDIR, "output-parent component is not a directory") + try: + child = os.open(part, flags, dir_fd=parent_descriptor) + except PermissionError: + # An ancestor may grant search but not read (a 0711 /home). + # Only the output parent itself must be listed, locked and + # synced. + if not _SEARCH_ONLY_FLAG or created or index == len(parts) - 1: + raise + child = os.open( + part, + _SEARCH_ONLY_FLAG | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=parent_descriptor, + ) + descriptors.append(child) + opened = os.fstat(child) + named = os.stat( + part, + dir_fd=parent_descriptor, + follow_symlinks=False, + ) + identity = (opened.st_dev, opened.st_ino) + if ( + not stat.S_ISDIR(opened.st_mode) + or identity != (expected.st_dev, expected.st_ino) + or identity != (named.st_dev, named.st_ino) + ): + raise OSError(errno.ESTALE, "output-parent component changed") + identities.append(identity) + descriptor = child + if created and first_created_index is None: + first_created_index = len(descriptors) - 1 + + # ``RetainedDirectory`` owns the final directory descriptor. The + # ancestor descriptors exist only long enough to create and fsync the + # path durably; close them before returning without surrendering the + # descriptor through which publication proceeds. + binding = RetainedDirectory(absolute, descriptors[-1], identities[-1]) + binding.verify() + if first_created_index is not None: + for containing_descriptor in reversed( + descriptors[first_created_index - 1 :] + ): + os.fsync(containing_descriptor) + binding.verify() + for ancestor_descriptor in reversed(descriptors[:-1]): + try: + os.close(ancestor_descriptor) + except OSError: + pass + descriptors = descriptors[-1:] + return binding + except BaseException: + if binding is not None: + binding.close() + descriptors = descriptors[:-1] + for descriptor in reversed(descriptors): + try: + os.close(descriptor) + except OSError: + pass + raise + + +def _create_workspace( + parent: Path, + parent_binding: RetainedDirectory, +) -> tuple[Path, tuple[int, int], tuple[tuple[int, ...], str]]: + try: + parent_binding.verify() + except OSError as error: + raise PublicationError(["output parent changed before staging"]) from error + parent_descriptor = parent_binding.descriptor + for _ in range(128): + name = f"{_PUBLICATION_STAGE_PREFIX}{secrets.token_hex(16)}" + workspace = parent / name + try: + os.mkdir(name, mode=0o700, dir_fd=parent_descriptor) + except FileExistsError: + continue + except OSError as error: + raise PublicationError( + [ + f"could not create a publication workspace in {parent}: " + f"{error.strerror or error}; render needs write access to the " + "output directory's parent" + ] + ) from error + identity: tuple[int, int] | None = None + marker: tuple[tuple[int, ...], str] | None = None + descriptor: int | None = None + try: + metadata = os.stat(name, dir_fd=parent_descriptor, follow_symlinks=False) + identity = metadata.st_dev, metadata.st_ino + descriptor = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=parent_descriptor, + ) + if _descriptor_identity(descriptor) != identity: + raise OSError(errno.ESTALE, "workspace changed during creation") + marker = _create_workspace_marker(descriptor) + os.fsync(parent_descriptor) + os.close(descriptor) + descriptor = None + return workspace, identity, marker + except BaseException as error: + if descriptor is not None: + try: + os.close(descriptor) + except OSError: + pass + descriptor = None + cleaned = bool( + identity is not None + and _remove_owned_workspace( + workspace, + identity, + expected_children={}, + expected_files=( + {_WORKSPACE_MARKER: marker} + if marker is not None + else {} + ), + parent_binding=parent_binding, + ) + ) + if cleaned: + raise + raise _PublicationRecoveryError( + [ + "publication workspace creation failed; workspace retained " + f"{_workspace_recovery_location(workspace, parent_binding)}" + ] + ) from error + finally: + if descriptor is not None: + try: + os.close(descriptor) + except OSError: + pass + raise PublicationError(["could not create a private publication workspace"]) + + +def _create_workspace_marker( + workspace_descriptor: int, +) -> tuple[tuple[int, ...], str]: + """Create and durably bind the marker that keeps workspaces out of Lean capture.""" + + descriptor: int | None = None + created_identity: tuple[int, int] | None = None + try: + descriptor = os.open( + _WORKSPACE_MARKER, + os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_CLOEXEC | os.O_NOFOLLOW, + 0o600, + dir_fd=workspace_descriptor, + ) + opened = os.fstat(descriptor) + created_identity = opened.st_dev, opened.st_ino + if not stat.S_ISREG(opened.st_mode): + raise OSError(errno.ESTALE, "workspace marker is not a regular file") + view = memoryview(_WORKSPACE_MARKER_BYTES) + written = 0 + while written < len(view): + try: + count = os.write(descriptor, view[written:]) + except InterruptedError: + continue + if count <= 0: + raise OSError(errno.EIO, "workspace marker write was incomplete") + written += count + os.fsync(descriptor) + final = _stat_signature(os.fstat(descriptor)) + named = _stat_signature( + os.stat( + _WORKSPACE_MARKER, + dir_fd=workspace_descriptor, + follow_symlinks=False, + ) + ) + if ( + final != named + or final[:2] != created_identity + ): + raise OSError(errno.ESTALE, "workspace marker changed during creation") + os.fsync(workspace_descriptor) + marker = final, hashlib.sha256(_WORKSPACE_MARKER_BYTES).hexdigest() + os.close(descriptor) + descriptor = None + return marker + except BaseException: + if created_identity is not None: + try: + current = os.stat( + _WORKSPACE_MARKER, + dir_fd=workspace_descriptor, + follow_symlinks=False, + ) + if (current.st_dev, current.st_ino) == created_identity: + os.unlink(_WORKSPACE_MARKER, dir_fd=workspace_descriptor) + except OSError: + pass + raise + finally: + if descriptor is not None: + try: + os.close(descriptor) + except OSError: + pass + + +def _open_workspace_directory( + parent_binding: RetainedDirectory, + name: str, + expected_identity: tuple[int, int], +) -> int: + _require_output_parent(parent_binding, "before opening the publication workspace") + descriptor = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=parent_binding.descriptor, + ) + try: + if _descriptor_identity(descriptor) != expected_identity: + raise PublicationError(["publication workspace changed before staging"]) + return descriptor + except BaseException: + os.close(descriptor) + raise + + +def _probe_publication_filesystem( + workspace: Path, + workspace_descriptor: int, + output_parent: RetainedDirectory, +) -> None: + """Exercise the commit primitives on this filesystem before live inspection. + + libc exposing ``renameat2`` or ``renameatx_np`` does not prove that the + mounted filesystem implements their no-replace and exchange flags. The + probe stays inside Autoform's private workspace and is removed before any + source or destination is inspected. + """ + + assert fcntl is not None + # Lock the private workspace, not the shared output parent: the probe only + # needs to know flock works on this filesystem, and must not contend with + # other renders. + try: + fcntl.flock(workspace_descriptor, fcntl.LOCK_EX | fcntl.LOCK_NB) + except OSError as error: + raise PublicationError( + ["transactional publication locking is unavailable on this filesystem"] + ) from error + + probe_name = ".capabilities" + probe_descriptor: int | None = None + peer_descriptor: int | None = None + probe_identity: tuple[int, int] | None = None + probe_created = False + operation_error: BaseException | None = None + cleanup_error: BaseException | None = None + + def named_directory_identity(parent: int, name: str) -> tuple[int, int]: + metadata = os.stat(name, dir_fd=parent, follow_symlinks=False) + if not stat.S_ISDIR(metadata.st_mode): + raise OSError(errno.ESTALE, "publication capability entry changed type") + return metadata.st_dev, metadata.st_ino + + def probe_rename( + source_parent: int, + source: str, + target_parent: int, + target: str, + *, + exchange: bool, + ) -> int: + function, flag = _rename_implementation(exchange=exchange) + result = function( + source_parent, + os.fsencode(source), + target_parent, + os.fsencode(target), + flag, + ) + return 0 if result == 0 else ctypes.get_errno() + + try: + os.mkdir(probe_name, mode=0o700, dir_fd=workspace_descriptor) + probe_created = True + probe_identity = named_directory_identity(workspace_descriptor, probe_name) + probe_descriptor = os.open( + probe_name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=workspace_descriptor, + ) + if _descriptor_identity(probe_descriptor) != probe_identity: + raise OSError(errno.ESTALE, "publication capability directory changed") + + os.mkdir("peer", mode=0o700, dir_fd=probe_descriptor) + peer_identity = named_directory_identity(probe_descriptor, "peer") + peer_descriptor = os.open( + "peer", + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=probe_descriptor, + ) + if _descriptor_identity(peer_descriptor) != peer_identity: + raise OSError(errno.ESTALE, "publication capability peer changed") + + os.mkdir("exchange-source", mode=0o700, dir_fd=probe_descriptor) + os.mkdir("exchange-target", mode=0o700, dir_fd=peer_descriptor) + source_identity = named_directory_identity(probe_descriptor, "exchange-source") + target_identity = named_directory_identity(peer_descriptor, "exchange-target") + exchange_error = probe_rename( + probe_descriptor, + "exchange-source", + peer_descriptor, + "exchange-target", + exchange=True, + ) + if exchange_error: + raise OSError( + exchange_error, + os.strerror(exchange_error), + "exchange-target", + ) + if ( + named_directory_identity(probe_descriptor, "exchange-source") + != target_identity + or named_directory_identity(peer_descriptor, "exchange-target") + != source_identity + ): + raise OSError(errno.ESTALE, "atomic directory exchange was not exact") + + os.mkdir("noreplace-source", mode=0o700, dir_fd=probe_descriptor) + os.mkdir("noreplace-collision", mode=0o700, dir_fd=peer_descriptor) + noreplace_identity = named_directory_identity(probe_descriptor, "noreplace-source") + collision_identity = named_directory_identity(peer_descriptor, "noreplace-collision") + collision_error = probe_rename( + probe_descriptor, + "noreplace-source", + peer_descriptor, + "noreplace-collision", + exchange=False, + ) + if collision_error not in {errno.EEXIST, errno.ENOTEMPTY}: + if collision_error: + raise OSError( + collision_error, + os.strerror(collision_error), + "noreplace-collision", + ) + raise OSError(errno.ENOTSUP, "no-replace rename overwrote an existing entry") + if ( + named_directory_identity(probe_descriptor, "noreplace-source") + != noreplace_identity + or named_directory_identity(peer_descriptor, "noreplace-collision") + != collision_identity + ): + raise OSError(errno.ESTALE, "no-replace collision changed an entry") + install_error = probe_rename( + probe_descriptor, + "noreplace-source", + peer_descriptor, + "noreplace-installed", + exchange=False, + ) + if install_error: + raise OSError( + install_error, + os.strerror(install_error), + "noreplace-installed", + ) + if ( + named_directory_identity(peer_descriptor, "noreplace-installed") + != noreplace_identity + ): + raise OSError(errno.ESTALE, "no-replace install changed generation") + os.fsync(peer_descriptor) + os.fsync(probe_descriptor) + os.fsync(workspace_descriptor) + except BaseException as error: + operation_error = error + finally: + if peer_descriptor is not None: + try: + os.close(peer_descriptor) + except OSError as error: + cleanup_error = error + if probe_descriptor is not None: + try: + inventory = _cleanup_inventory_descriptor(probe_descriptor) + _remove_inventory_contents(probe_descriptor, inventory) + except BaseException as error: + cleanup_error = cleanup_error or error + try: + os.close(probe_descriptor) + except OSError as error: + cleanup_error = cleanup_error or error + elif probe_created: + cleanup_error = cleanup_error or OSError( + errno.ESTALE, + "publication capability directory could not be retained", + ) + if probe_identity is not None and cleanup_error is None: + try: + if ( + named_directory_identity(workspace_descriptor, probe_name) + != probe_identity + ): + raise OSError( + errno.ESTALE, + "publication capability directory changed during cleanup", + ) + os.rmdir(probe_name, dir_fd=workspace_descriptor) + os.fsync(workspace_descriptor) + except BaseException as error: + cleanup_error = error + try: + fcntl.flock(workspace_descriptor, fcntl.LOCK_UN) + except BaseException as error: + cleanup_error = cleanup_error or error + + if cleanup_error is not None: + raise _PublicationRecoveryError( + [ + "publication capability probe could not be cleaned; workspace retained " + f"{_workspace_recovery_location(workspace, output_parent)}" + ] + ) from (operation_error or cleanup_error) + if operation_error is not None: + if isinstance(operation_error, (KeyboardInterrupt, SystemExit)): + raise operation_error + raise PublicationError( + ["transactional publication is unavailable on this filesystem"] + ) from operation_error + + +def _create_stage_directory(workspace_descriptor: int) -> tuple[int, tuple[int, int]]: + created_identity: tuple[int, int] | None = None + descriptor: int | None = None + try: + # The umask decides the published modes, as plain writes did on main; + # the 0700 workspace keeps the stage private until the commit. + os.mkdir("site", mode=0o777, dir_fd=workspace_descriptor) + metadata = os.stat("site", dir_fd=workspace_descriptor, follow_symlinks=False) + created_identity = metadata.st_dev, metadata.st_ino + descriptor = os.open( + "site", + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=workspace_descriptor, + ) + identity = _descriptor_identity(descriptor) + if identity != created_identity: + raise PublicationError(["publication stage changed during creation"]) + return descriptor, identity + except BaseException: + if descriptor is not None: + try: + os.close(descriptor) + except OSError: + pass + if created_identity is not None: + try: + current = os.stat( + "site", + dir_fd=workspace_descriptor, + follow_symlinks=False, + ) + if ( + stat.S_ISDIR(current.st_mode) + and (current.st_dev, current.st_ino) == created_identity + ): + cleanup_descriptor = os.open( + "site", + os.O_RDONLY + | os.O_CLOEXEC + | os.O_DIRECTORY + | os.O_NOFOLLOW, + dir_fd=workspace_descriptor, + ) + try: + if ( + _descriptor_identity(cleanup_descriptor) + == created_identity + and _directory_names_match(cleanup_descriptor, ()) + ): + os.rmdir("site", dir_fd=workspace_descriptor) + finally: + os.close(cleanup_descriptor) + except OSError: + pass + raise + + +def _remove_owned_workspace( + workspace: Path, + identity: tuple[int, int], + *, + expected_children: dict[str, dict[tuple[int, int], _CleanupInventory]], + expected_files: Mapping[str, tuple[tuple[int, ...], str]] | None = None, + parent_binding: RetainedDirectory | None = None, +) -> bool: + """Remove only inventoried trees in Autoform's random mode-0700 workspace. + + The inventory and atomic per-file claim protect against ordinary concurrent + path replacement. POSIX has no unlink-if-inode operation, so code running + as the same user must not be given a path or callback into this private + workspace while cleanup is active. + """ + + parent_descriptor: int | None = None + workspace_descriptor: int | None = None + close_parent = False + try: + if parent_binding is None: + parent_descriptor = _open_directory_path(workspace.parent) + close_parent = True + else: + if not _output_parent_is_current(parent_binding): + return False + parent_descriptor = parent_binding.descriptor + try: + metadata = os.stat( + workspace.name, dir_fd=parent_descriptor, follow_symlinks=False + ) + except FileNotFoundError: + return False + if not stat.S_ISDIR(metadata.st_mode) or ( + metadata.st_dev, + metadata.st_ino, + ) != identity: + return False + workspace_descriptor = os.open( + workspace.name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=parent_descriptor, + ) + if _descriptor_identity(workspace_descriptor) != identity: + return False + files = dict(expected_files or {}) + names = set( + _bounded_directory_names( + workspace_descriptor, + budget=_InventoryBudget(), + depth=1, + ) + ) + if names != set(expected_children) | set(files): + return False + for name in files: + metadata = os.stat( + name, + dir_fd=workspace_descriptor, + follow_symlinks=False, + ) + expected_identity, expected_digest = files[name] + if ( + not stat.S_ISREG(metadata.st_mode) + or _stat_signature(metadata) != expected_identity + ): + return False + data = _read_regular_file_at( + workspace_descriptor, + name, + workspace / name, + max_bytes=_PUBLICATION_MAX_FILE_BYTES, + ignore_close_errors=True, + ) + if ( + hashlib.sha256(data).hexdigest() != expected_digest + or _stat_signature( + os.stat( + name, + dir_fd=workspace_descriptor, + follow_symlinks=False, + ) + ) + != expected_identity + ): + return False + for name in expected_children: + child = os.stat(name, dir_fd=workspace_descriptor, follow_symlinks=False) + child_identity = (child.st_dev, child.st_ino) + expected_inventory = expected_children[name].get(child_identity) + if not stat.S_ISDIR(child.st_mode) or expected_inventory is None: + return False + child_descriptor = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=workspace_descriptor, + ) + try: + if _descriptor_identity(child_descriptor) != child_identity: + return False + if _cleanup_inventory_descriptor(child_descriptor) != expected_inventory: + return False + finally: + os.close(child_descriptor) + for name in sorted(expected_children): + current = os.stat(name, dir_fd=workspace_descriptor, follow_symlinks=False) + child_identity = (current.st_dev, current.st_ino) + expected_inventory = expected_children[name].get(child_identity) + if not stat.S_ISDIR(current.st_mode) or expected_inventory is None: + return False + child_descriptor = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=workspace_descriptor, + ) + try: + _remove_inventory_contents( + child_descriptor, + expected_inventory, + ) + finally: + os.close(child_descriptor) + os.rmdir(name, dir_fd=workspace_descriptor) + if files: + _remove_inventory_contents( + workspace_descriptor, + _CleanupInventory( + (), + tuple( + sorted( + (name, identity, digest) + for name, (identity, digest) in files.items() + ) + ), + ), + ) + current = os.stat( + workspace.name, dir_fd=parent_descriptor, follow_symlinks=False + ) + if (current.st_dev, current.st_ino) != identity: + return False + os.rmdir(workspace.name, dir_fd=parent_descriptor) + return True + except (OSError, PublicationError): + return False + finally: + if workspace_descriptor is not None: + try: + os.close(workspace_descriptor) + except OSError: + pass + if close_parent and parent_descriptor is not None: + try: + os.close(parent_descriptor) + except OSError: + pass + + +def _cleanup_inventory( + root: Path, + *, + expected_identity: tuple[int, int] | None = None, +) -> _CleanupInventory: + descriptor = _open_directory_path(root) + try: + if expected_identity is not None and _descriptor_identity(descriptor) != expected_identity: + raise PublicationError(["publication tree changed before cleanup inventory"]) + return _cleanup_inventory_descriptor(descriptor) + finally: + os.close(descriptor) + + +def _bounded_directory_names( + descriptor: int, + *, + budget: _InventoryBudget, + depth: int, +) -> tuple[str, ...]: + names: list[str] = [] + try: + with os.scandir(descriptor) as iterator: + for entry in iterator: + budget.add_entry(depth=depth) + names.append(entry.name) + except (OSError, TypeError) as error: + raise PublicationError( + ["bounded publication traversal is unavailable on this platform"] + ) from error + return tuple(sorted(names)) + + +def _directory_names_match(descriptor: int, expected: tuple[str, ...]) -> bool: + names: list[str] = [] + with os.scandir(descriptor) as iterator: + for entry in iterator: + if len(names) == len(expected): + return False + names.append(entry.name) + return tuple(sorted(names)) == expected + + +def _cleanup_inventory_descriptor(descriptor: int) -> _CleanupInventory: + directories: list[tuple[str, tuple[int, ...]]] = [] + files: list[tuple[str, tuple[int, ...], str]] = [] + budget = _InventoryBudget() + + def visit(current: int, prefix: str, depth: int) -> None: + names = _bounded_directory_names( + current, + budget=budget, + depth=depth + 1, + ) + for name in names: + relative = f"{prefix}/{name}" if prefix else name + metadata = os.stat(name, dir_fd=current, follow_symlinks=False) + identity = _stat_signature(metadata) + if stat.S_ISDIR(metadata.st_mode): + directories.append((relative, identity)) + child = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=current, + ) + try: + if _stat_signature(os.fstat(child)) != identity: + raise OSError(errno.ESTALE, "workspace changed during cleanup inventory") + visit(child, relative, depth + 1) + finally: + os.close(child) + if _stat_signature( + os.stat(name, dir_fd=current, follow_symlinks=False) + ) != identity: + raise OSError(errno.ESTALE, "workspace changed during cleanup inventory") + continue + if not stat.S_ISREG(metadata.st_mode): + raise OSError(errno.ESTALE, "workspace contains an unowned filesystem entry") + budget.add_file(metadata.st_size) + data = _read_regular_file_at( + current, + name, + Path(relative), + max_bytes=_PUBLICATION_MAX_FILE_BYTES, + ) + after = os.stat(name, dir_fd=current, follow_symlinks=False) + if _stat_signature(after) != identity: + raise OSError(errno.ESTALE, "workspace changed during cleanup inventory") + files.append((relative, identity, hashlib.sha256(data).hexdigest())) + if not _directory_names_match(current, names): + raise OSError(errno.ESTALE, "workspace changed during cleanup inventory") + + visit(descriptor, "", 0) + return _CleanupInventory(tuple(sorted(directories)), tuple(sorted(files))) + + +def _remove_inventory_contents( + descriptor: int, + inventory: _CleanupInventory, + *, + prefix: str = "", + budget: _InventoryBudget | None = None, + index: _CleanupIndex | None = None, + depth: int = 0, +) -> None: + if budget is None: + budget = _InventoryBudget() + if index is None: + index = _index_cleanup_inventory(inventory) + expected_names = set(index.children.get(prefix, ())) + names = set( + _bounded_directory_names( + descriptor, + budget=budget, + depth=depth + 1, + ) + ) + if names != expected_names: + raise OSError(errno.ESTALE, "workspace changed during cleanup") + for name in sorted(names): + relative = f"{prefix}/{name}" if prefix else name + metadata = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + expected_directory = index.directories.get(relative) + if expected_directory is not None: + if _stat_signature(metadata) != expected_directory: + raise OSError(errno.ESTALE, "workspace changed during cleanup") + child = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=descriptor, + ) + try: + if _stat_signature(os.fstat(child)) != expected_directory: + raise OSError(errno.ESTALE, "workspace changed during cleanup") + _remove_inventory_contents( + child, + inventory, + prefix=relative, + budget=budget, + index=index, + depth=depth + 1, + ) + finally: + os.close(child) + current = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + if not stat.S_ISDIR(current.st_mode) or ( + current.st_dev, + current.st_ino, + ) != (metadata.st_dev, metadata.st_ino): + raise OSError(errno.ESTALE, "workspace changed during cleanup") + os.rmdir(name, dir_fd=descriptor) + continue + expected_file = index.files.get(relative) + if expected_file is None or _stat_signature(metadata) != expected_file[0]: + raise OSError(errno.ESTALE, "workspace changed during cleanup") + quarantine = f".autoform-cleanup-{secrets.token_hex(16)}" + try: + _cleanup_rename_noreplace(descriptor, name, quarantine) + except OSError as error: + if error.errno not in _MISSING_RENAME_FLAG_ERRORS: + raise + # NFS, 9p and sshfs reject every rename flag. The capability probe + # refuses them before anything is staged, so this only removes the + # workspace marker. Verify it in place and unlink it. + data = _read_regular_file_at( + descriptor, + name, + Path(relative), + max_bytes=_PUBLICATION_MAX_FILE_BYTES, + ignore_close_errors=True, + ) + current = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + if ( + _stat_signature(current) != expected_file[0] + or hashlib.sha256(data).hexdigest() != expected_file[1] + ): + raise OSError(errno.ESTALE, "workspace changed during cleanup") from error + os.unlink(name, dir_fd=descriptor) + continue + try: + claimed = os.stat( + quarantine, + dir_fd=descriptor, + follow_symlinks=False, + ) + data = _read_regular_file_at( + descriptor, + quarantine, + Path(relative), + max_bytes=_PUBLICATION_MAX_FILE_BYTES, + ignore_close_errors=True, + ) + final = os.stat( + quarantine, + dir_fd=descriptor, + follow_symlinks=False, + ) + if ( + _stat_signature(claimed) != expected_file[0] + or _stat_signature(final) != expected_file[0] + or hashlib.sha256(data).hexdigest() != expected_file[1] + ): + raise OSError(errno.ESTALE, "workspace changed during cleanup") + except BaseException: + try: + _cleanup_rename_noreplace(descriptor, quarantine, name) + except BaseException: + pass + raise + os.unlink(quarantine, dir_fd=descriptor) + if not _directory_names_match(descriptor, ()): + raise OSError(errno.ESTALE, "workspace changed during cleanup") + + +def _index_cleanup_inventory(inventory: _CleanupInventory) -> _CleanupIndex: + """Index each cleanup record once by path and immediate parent.""" + + directories: dict[str, tuple[int, ...]] = {} + files: dict[str, tuple[tuple[int, ...], str]] = {} + children: dict[str, set[str]] = {} + + def add_path(relative: str) -> None: + if not _valid_inventory_path(relative): + raise PublicationError(["cleanup inventory contains an invalid path"]) + parent, separator, name = relative.rpartition("/") + if not separator: + parent = "" + name = relative + children.setdefault(parent, set()).add(name) + + for relative, identity in inventory.directories: + if relative in directories or relative in files: + raise PublicationError(["cleanup inventory contains duplicate paths"]) + add_path(relative) + directories[relative] = identity + for relative, identity, digest in inventory.files: + if relative in directories or relative in files: + raise PublicationError(["cleanup inventory contains duplicate paths"]) + add_path(relative) + files[relative] = identity, digest + for paths in (directories, files): + for relative in paths: + parent = relative.rpartition("/")[0] + if parent and parent not in directories: + raise PublicationError( + ["cleanup inventory has a missing parent directory"] + ) + return _CleanupIndex( + directories, + files, + {parent: tuple(sorted(names)) for parent, names in children.items()}, + ) + + +_MISSING_RENAME_FLAG_ERRORS = frozenset( + {errno.EINVAL, errno.ENOSYS, errno.ENOTSUP, errno.EOPNOTSUPP} +) + + +def _cleanup_rename_noreplace(descriptor: int, source: str, target: str) -> None: + """Claim a cleanup entry without using the publication commit hooks.""" + + function, flag = _rename_implementation(exchange=False) + result = function( + descriptor, + os.fsencode(source), + descriptor, + os.fsencode(target), + flag, + ) + if result != 0: + error = ctypes.get_errno() + raise OSError(error, os.strerror(error), target) + + +def _require_destination_inventory( + inventory: _CleanupInventory, + state: _DestinationState, + *, + allow_os_metadata: bool = False, +) -> None: + directories = tuple(path for path, _identity in inventory.directories) + files = tuple( + (path, digest) + for path, _identity, digest in inventory.files + if path != PUBLICATION_MANIFEST + and not (allow_os_metadata and _is_os_metadata(path)) + ) + manifests = tuple( + digest + for path, _identity, digest in inventory.files + if path == PUBLICATION_MANIFEST + ) + if state.kind == "empty": + valid = not directories and not files and not manifests + else: + valid = ( + state.kind == "owned" + and directories == state.directories + and files == state.files + and manifests == (state.manifest_sha256,) + ) + if not valid: + raise PublicationError(["publication tree changed before cleanup inventory"]) + + +def _parse_inventory_files(value: object) -> tuple[tuple[str, str], ...]: + if not isinstance(value, dict): + raise PublicationError(["publication manifest has no valid file inventory"]) + files: list[tuple[str, str]] = [] + for path, digest in value.items(): + if ( + not isinstance(path, str) + or not _valid_inventory_path(path) + or path == PUBLICATION_MANIFEST + or not isinstance(digest, str) + or re.fullmatch(r"[0-9a-f]{64}", digest) is None + ): + raise PublicationError(["publication manifest has an invalid file inventory"]) + files.append((path, digest)) + if files != sorted(files): + raise PublicationError(["publication manifest file inventory is not canonical"]) + return tuple(files) + + +def _parse_inventory_directories(value: object) -> tuple[str, ...]: + if not isinstance(value, list): + raise PublicationError(["publication manifest has no valid directory inventory"]) + if any(not isinstance(path, str) or not _valid_inventory_path(path) for path in value): + raise PublicationError(["publication manifest has an invalid directory inventory"]) + if value != sorted(set(value)): + raise PublicationError(["publication manifest directory inventory is not canonical"]) + return tuple(value) + + +def _valid_inventory_path(value: str) -> bool: + path = PurePosixPath(value) + return ( + bool(value) + and value != "." + and "\x00" not in value + and "\\" not in value + and not path.is_absolute() + and path.as_posix() == value + and ".." not in path.parts + ) + + +def _publication_inventory_descriptor( + root_descriptor: int, +) -> tuple[tuple[str, ...], tuple[tuple[str, str], ...]]: + directories: list[str] = [] + files: list[tuple[str, str]] = [] + budget = _InventoryBudget() + + def visit(descriptor: int, prefix: str, depth: int) -> None: + try: + names = _bounded_directory_names( + descriptor, + budget=budget, + depth=depth + 1, + ) + for name in names: + relative = f"{prefix}/{name}" if prefix else name + metadata = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + if stat.S_ISLNK(metadata.st_mode): + raise PublicationError( + [f"refusing symlink in publication output: {relative}"] + ) + if stat.S_ISDIR(metadata.st_mode): + directories.append(relative) + child = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=descriptor, + ) + try: + if _descriptor_identity(child) != (metadata.st_dev, metadata.st_ino): + raise PublicationError( + ["publication output changed while it was inspected"] + ) + visit(child, relative, depth + 1) + finally: + os.close(child) + current = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + if (current.st_dev, current.st_ino) != (metadata.st_dev, metadata.st_ino): + raise PublicationError( + ["publication output changed while it was inspected"] + ) + continue + if not stat.S_ISREG(metadata.st_mode): + raise PublicationError( + [f"refusing non-regular publication output: {relative}"] + ) + budget.add_file(metadata.st_size) + if relative == PUBLICATION_MANIFEST: + continue + data = _read_regular_file_at( + descriptor, + name, + Path(relative), + max_bytes=_PUBLICATION_MAX_FILE_BYTES, + ) + files.append((relative, hashlib.sha256(data).hexdigest())) + if not _directory_names_match(descriptor, names): + raise PublicationError( + ["publication output changed while it was inspected"] + ) + except OSError as error: + raise PublicationError( + ["publication output changed while it was inspected"] + ) from error + + visit(root_descriptor, "", 0) + return tuple(sorted(directories)), tuple(sorted(files)) + + +def _read_owned_publication_files_at( + parent_descriptor: int, + name: str, + state: _DestinationState, +) -> dict[PurePosixPath, bytes]: + """Read an exact prior generation through its retained parent descriptor.""" + + if state.kind != "owned" or state.identity is None: + raise PublicationError(["cannot seed from an unowned publication"]) + descriptor = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=parent_descriptor, + ) + try: + if _descriptor_identity(descriptor) != state.identity: + raise PublicationError(["publication output changed while it was copied"]) + copied: dict[PurePosixPath, bytes] = {} + for relative, expected_digest in state.files: + path = PurePosixPath(relative) + data = _read_relative_regular_file(descriptor, path) + if hashlib.sha256(data).hexdigest() != expected_digest: + raise PublicationError(["publication output changed while it was copied"]) + copied[path] = data + return copied + finally: + os.close(descriptor) + + +def _read_relative_regular_file( + root_descriptor: int, + relative: PurePosixPath, +) -> bytes: + descriptor = os.dup(root_descriptor) + try: + for part in relative.parts[:-1]: + child = os.open( + part, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=descriptor, + ) + os.close(descriptor) + descriptor = child + return _read_regular_file_at( + descriptor, + relative.name, + Path(relative.as_posix()), + max_bytes=_PUBLICATION_MAX_FILE_BYTES, + ) + finally: + os.close(descriptor) + + +def _publication_plan_checkpoint(_event: str, _relative: str) -> None: + """A test hook for adversarial pathname replacement.""" + + +def _publication_commit_checkpoint(_event: str) -> None: + """A test hook for process termination immediately after the atomic commit.""" + + +def _materialize_publication_plan( + stage_descriptor: int, + plan: _PublicationFilePlan, +) -> None: + """Write a validated plan only through retained, no-follow descriptors.""" + + plan.inventory() + tree: dict[str, object] = {"directories": {}, "files": {}} + for relative, data in plan.files.items(): + node = tree + for part in relative.parts[:-1]: + directories = node["directories"] + assert isinstance(directories, dict) + node = directories.setdefault(part, {"directories": {}, "files": {}}) + assert isinstance(node, dict) + files = node["files"] + assert isinstance(files, dict) + files[relative.name] = data + + + def write_node(descriptor: int, node: dict[str, object], prefix: str) -> None: + directories = node["directories"] + files = node["files"] + assert isinstance(directories, dict) + assert isinstance(files, dict) + for name in sorted(directories): + relative = f"{prefix}/{name}" if prefix else name + os.mkdir(name, mode=0o777, dir_fd=descriptor) + metadata = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + child = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=descriptor, + ) + try: + if _descriptor_identity(child) != (metadata.st_dev, metadata.st_ino): + raise PublicationError(["publication stage changed during materialization"]) + _publication_plan_checkpoint("after-directory-open", relative) + child_node = directories[name] + assert isinstance(child_node, dict) + write_node(child, child_node, relative) + finally: + os.close(child) + for name in sorted(files): + relative = f"{prefix}/{name}" if prefix else name + file_descriptor = os.open( + name, + os.O_WRONLY + | os.O_CREAT + | os.O_EXCL + | os.O_CLOEXEC + | os.O_NOFOLLOW + | os.O_NONBLOCK, + 0o666, + dir_fd=descriptor, + ) + try: + metadata = os.fstat(file_descriptor) + if not stat.S_ISREG(metadata.st_mode) or metadata.st_size != 0: + raise PublicationError(["publication stage file was not created safely"]) + _publication_plan_checkpoint("after-file-open", relative) + data = files[name] + assert isinstance(data, bytes) + view = memoryview(data) + while view: + written = os.write(file_descriptor, view) + if written <= 0: + raise OSError(errno.EIO, "publication stage write made no progress") + view = view[written:] + if os.fstat(file_descriptor).st_size != len(data): + raise OSError(errno.EIO, "publication stage write was incomplete") + finally: + os.close(file_descriptor) + + write_node(stage_descriptor, tree, "") + + +def _update_source_digest( + digest, + relative: Path, + data: bytes, + *, + kind: bytes = b"file", +) -> None: + path = os.fsencode(relative.as_posix()) + digest.update(len(kind).to_bytes(8, "big")) + digest.update(kind) + digest.update(len(path).to_bytes(8, "big")) + digest.update(path) + digest.update(len(data).to_bytes(8, "big")) + digest.update(data) + + +def _read_regular_file_at( + parent_descriptor: int, + name: str, + display_path: Path, + *, + max_bytes: int, + ignore_close_errors: bool = False, +) -> bytes: + flags = os.O_RDONLY | os.O_CLOEXEC | os.O_NOFOLLOW | os.O_NONBLOCK + try: + descriptor = os.open(name, flags, dir_fd=parent_descriptor) + except OSError as error: + raise PublicationError( + [f"could not safely read regular file: {display_path.name}"] + ) from error + try: + before = os.fstat(descriptor) + if not stat.S_ISREG(before.st_mode): + raise PublicationError([f"refusing non-regular file: {display_path.name}"]) + if before.st_size > max_bytes: + raise PublicationError( + [f"publication file exceeds max_file_bytes={max_bytes}: {display_path}"] + ) + chunks: list[bytes] = [] + remaining = max_bytes + 1 + while True: + chunk = os.read(descriptor, min(1024 * 1024, remaining)) + if not chunk: + break + chunks.append(chunk) + remaining -= len(chunk) + if remaining == 0: + raise PublicationError( + [f"publication file exceeds max_file_bytes={max_bytes}: {display_path}"] + ) + after = os.fstat(descriptor) + if ( + (before.st_dev, before.st_ino, before.st_size, before.st_mtime_ns, before.st_ctime_ns) + != (after.st_dev, after.st_ino, after.st_size, after.st_mtime_ns, after.st_ctime_ns) + ): + raise PublicationError([f"file changed while it was read: {display_path.name}"]) + entry = os.stat(name, dir_fd=parent_descriptor, follow_symlinks=False) + if (entry.st_dev, entry.st_ino) != (after.st_dev, after.st_ino): + raise PublicationError([f"file changed while it was read: {display_path.name}"]) + return b"".join(chunks) + finally: + try: + os.close(descriptor) + except OSError: + if not ignore_close_errors: + raise + + +_LEAN_CAPTURE_ATTEMPTS = 3 +_LEAN_CAPTURE_RETRY_SECONDS = 0.05 + + +def _open_lean_sources( + root: Path, *, exclude_roots: Iterable[Path] +) -> BoundProjectSources: + return open_project_sources( + root, + exclude_roots=exclude_roots, + limits=_LEAN_CAPTURE_LIMITS, + opaque_markers=( + OpaqueDirectoryMarker( + PUBLICATION_MANIFEST, + PUBLICATION_MANIFEST_MAX_BYTES, + _is_complete_publication_manifest, + ), + OpaqueDirectoryMarker( + _WORKSPACE_MARKER, + len(_WORKSPACE_MARKER_BYTES), + _is_publication_workspace_marker, + ), + ), + ) + + +def _bind_lean_sources( + root: Path, *, exclude_roots: Iterable[Path] +) -> tuple[BoundProjectSources, IndexedSourceSnapshot]: + """Bind one stable Lean generation, retrying a concurrent directory change. + + The walk verifies every listing it descends into, so unrelated churn under + the Lean root (editor swap files, logs) can interrupt it; main retried the + same way through ``lean.bind_project_source_snapshot``. + """ + + exclusions = tuple(exclude_roots) + for attempt in range(_LEAN_CAPTURE_ATTEMPTS): + if attempt: + time.sleep(_LEAN_CAPTURE_RETRY_SECONDS * attempt) + sources: BoundProjectSources | None = None + try: + sources = _open_lean_sources(root, exclude_roots=exclusions) + return sources, sources.capture() + except TreeChangedError: + if sources is not None: + sources.close() + continue + except BaseException as error: + if sources is not None: + sources.close() + if isinstance(error, TreeCaptureLimitError): + raise PublicationError([f"Lean source {error}"]) from error + if isinstance(error, OSError): + raise PublicationError([index_failure_message(error)]) from error + if isinstance(error, TreeSnapshotError): + raise PublicationError( + [f"Lean sources could not be indexed: {error}"] + ) from error + raise + if not root.is_dir(): + raise PublicationError( + ["Lean sources could not be indexed: the Lean root is not a directory"] + ) + raise PublicationError( + ["Lean sources kept changing while they were captured; retry"] + ) + + +def _is_complete_publication_manifest(data: bytes) -> bool: + if len(data) > PUBLICATION_MANIFEST_MAX_BYTES: + return False + try: + value = json.loads(data.decode("utf-8")) + except (UnicodeError, ValueError, RecursionError): + return False + return ( + isinstance(value, dict) + and value.get("schema") in _PUBLICATION_OPAQUE_SCHEMAS + and value.get("complete") is True + ) + + +def _is_publication_workspace_marker(data: bytes) -> bool: + return data == _WORKSPACE_MARKER_BYTES + + +def _verified_repository_ref( + repo_root: Path, + requested_ref: str, + *, + blueprint: _CapturedBlueprint, + lean_snapshot: IndexedSourceSnapshot, + reasons: list[str] | None = None, +) -> str | None: + """Return an immutable commit only when it contains the captured inputs. + + Source locations and vault links are derived from in-memory snapshots. A + Git URL is truthful only if the exact bytes used for those derivations are + blobs in the named commit; a stable dirty tree or an untracked note must + therefore fall back to local publication rather than borrow ``HEAD``. + """ + + files = [ + (blueprint.path(relative), data) + for relative, data in blueprint.files.items() + ] + files.extend( + (lean_snapshot.index.root / relative, data) + for relative, data in lean_snapshot.source_files + ) + sources = PurePosixPath(SOURCES_DIR) + required_directories = ( + (blueprint.path(sources),) if sources in blueprint.directories else () + ) + return verify_repository_snapshot( + repo_root, + requested_ref, + files, + required_directories=required_directories, + reasons=reasons, + ) + + +def _require_bound_lean_source_revision( + sources: BoundProjectSources, + expected_generation: str, +) -> None: + changed: TreeChangedError | None = None + for attempt in range(_LEAN_CAPTURE_ATTEMPTS): + if attempt: + time.sleep(_LEAN_CAPTURE_RETRY_SECONDS * attempt) + try: + sources.verify() + actual = sources.capture().generation_revision + sources.verify() + break + except TreeChangedError as error: + # Unrelated entries can change a listing without changing any + # Lean file; only the Lean generation below decides. + changed = error + except TreeCaptureLimitError as error: + raise PublicationError([f"Lean source {error}"]) from error + except (OSError, TreeSnapshotError) as error: + raise PublicationError( + ["Lean sources changed during publication; previous site was preserved"] + ) from error + else: + raise PublicationError( + ["Lean sources changed during publication; previous site was preserved"] + ) from changed + if actual != expected_generation: + raise PublicationError( + ["Lean sources changed during publication; previous site was preserved"] + ) + + +def _sync_tree_descriptor(root_descriptor: int) -> None: + """Sync a staged generation without resolving its pathname.""" - _prepare_destination(destination, clean=clean) - _write_publication_manifest( - destination, - blueprint, - graph, - linker, - coverage=coverage, - complete=False, - ) + budget = _InventoryBudget() - report = RenderReport(output_dir=destination) - node_paths = {node.path.resolve(): node for node in graph.nodes.values()} - # Nodes are published as environments on their milestone page, the way a - # blueprint chapter carries many statements in sequence. Each keeps an - # anchor so every cross-reference still lands on the statement itself. - groups = _group_nodes(graph) - containers = _containers(graph) - anchors = { - node_id: _anchor(node_id, group) - for group, node_ids in groups.items() - for node_id in node_ids - } - group_pages = {group: destination / _group_page(group) for group in groups} - targets = { - node_id: (group_pages[group], anchors[node_id]) - for group, node_ids in groups.items() - for node_id in node_ids - } - targets.update( - { - node_id: (destination / node.path.relative_to(blueprint), "") - for node_id, node in graph.nodes.items() - if node_id in containers or not node.formalizable - } + def visit(descriptor: int, prefix: str, depth: int) -> None: + names = _bounded_directory_names( + descriptor, + budget=budget, + depth=depth + 1, + ) + for name in names: + relative = f"{prefix}/{name}" if prefix else name + metadata = os.stat(name, dir_fd=descriptor, follow_symlinks=False) + signature = _stat_signature(metadata) + if stat.S_ISDIR(metadata.st_mode): + child = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=descriptor, + ) + try: + if _stat_signature(os.fstat(child)) != signature: + raise PublicationError( + ["staged publication changed while syncing"] + ) + visit(child, relative, depth + 1) + finally: + os.close(child) + if _stat_signature( + os.stat(name, dir_fd=descriptor, follow_symlinks=False) + ) != signature: + raise PublicationError( + ["staged publication changed while syncing"] + ) + continue + if not stat.S_ISREG(metadata.st_mode): + raise PublicationError( + [f"refusing non-regular staged publication entry: {relative}"] + ) + budget.add_file(metadata.st_size) + file_descriptor = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_NOFOLLOW | os.O_NONBLOCK, + dir_fd=descriptor, + ) + try: + if _stat_signature(os.fstat(file_descriptor)) != signature: + raise PublicationError( + ["staged publication changed while syncing"] + ) + os.fsync(file_descriptor) + if _stat_signature(os.fstat(file_descriptor)) != signature: + raise PublicationError( + ["staged publication changed while syncing"] + ) + finally: + os.close(file_descriptor) + if _stat_signature( + os.stat(name, dir_fd=descriptor, follow_symlinks=False) + ) != signature: + raise PublicationError(["staged publication changed while syncing"]) + if not _directory_names_match(descriptor, names): + raise PublicationError(["staged publication changed while syncing"]) + os.fsync(descriptor) + + visit(root_descriptor, "", 0) + + +def _publication_inventory_at( + parent_descriptor: int, + name: str, + expected_identity: tuple[int, int], +) -> _CleanupInventory: + descriptor = os.open( + name, + os.O_RDONLY | os.O_CLOEXEC | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=parent_descriptor, ) - node_sources = { - node.path.resolve(): node_id for node_id, node in graph.nodes.items() - } + try: + if _descriptor_identity(descriptor) != expected_identity: + raise PublicationError(["publication tree changed while it was inspected"]) + return _cleanup_inventory_descriptor(descriptor) + finally: + os.close(descriptor) - for source in sorted(blueprint.rglob("*")): - relative = source.relative_to(blueprint) - if _SKIPPED_DIRECTORIES.intersection(relative.parts) or _is_hidden(relative): - continue - if relative.name in _GENERATED_FILES: - continue - # Source notes leave the site entirely once readers can reach them in - # the repository, so the book has one reference surface rather than two. - if sources_base is not None and relative.parts[:1] == (SOURCES_DIR,): - continue - target = destination / relative - # Directories are created on demand below, so a directory holding - # nothing but absorbed nodes leaves no empty shell behind. - if source.is_dir(): - continue - # Narrative articles remain book pages. Only formalizable leaves are - # consolidated into their containing article with stable anchors. - article = node_paths.get(source.resolve()) - if article is not None and article.formalizable and article.id not in containers: - continue - target.parent.mkdir(parents=True, exist_ok=True) - if source.suffix.lower() == ".md": - rewritten = _rewrite_links( - source.read_text(encoding="utf-8"), - source_dir=source.parent, - page=target, - blueprint=blueprint, - destination=destination, - node_sources=node_sources, - targets=targets, - sources_base=sources_base, + +def _publish_staged_site( + stage: Path, + destination: Path, + expected: _DestinationState, + staged: _DestinationState, + *, + expected_inventory: _CleanupInventory | None, + staged_inventory: _CleanupInventory, + commit_state: _PublicationCommitState, + input_guard: Callable[[], None], + output_parent: RetainedDirectory, + workspace_identity: tuple[int, int], + workspace_descriptor: int, +) -> None: + """Commit one verified stage, without ever moving it back into place.""" + + stage_parent_descriptor = workspace_descriptor + try: + _require_output_parent(output_parent, "before publication commit") + parent_descriptor = output_parent.descriptor + if _descriptor_identity(stage_parent_descriptor) != workspace_identity: + raise _PublicationRecoveryError( + [ + "publication workspace changed; workspace retained " + f"{_workspace_recovery_location(stage.parent, output_parent)}" + ] ) - target.write_text(rewritten, encoding="utf-8") - else: - shutil.copy2(source, target) - report.pages += 1 + if os.fstat(parent_descriptor).st_dev != os.fstat(stage_parent_descriptor).st_dev: + raise PublicationError( + ["publication staging and output directories are on different filesystems"] + ) + if staged.kind != "owned" or staged.identity is None: + raise _PublicationRecoveryError( + [ + "publication stage changed; workspace retained " + f"{_workspace_recovery_location(stage.parent, output_parent)}" + ] + ) + _lock_output_parent(parent_descriptor, destination.parent, expected) - overview = destination / "README.md" - if overview.is_file(): - overview.write_text( - _render_landing_page( - overview.read_text(encoding="utf-8"), - graph=graph, - statuses=statuses, - coverage=coverage, - groups=groups, - group_pages=group_pages, - page=overview, - destination=destination, - ), - encoding="utf-8", + current = _inspect_destination_at( + parent_descriptor, + destination.name, + destination, + allow_os_metadata=True, ) + if current != expected: + raise PublicationError( + ["output directory changed during publication; previous site was preserved"] + ) + if expected.kind == "absent": + if expected_inventory is not None: + raise PublicationError(["invalid absent-destination publication state"]) + else: + if expected.identity is None or expected_inventory is None: + raise PublicationError(["invalid destination publication state"]) + if ( + _publication_inventory_at( + parent_descriptor, + destination.name, + expected.identity, + ) + != expected_inventory + ): + raise PublicationError( + ["output directory changed during publication; previous site was preserved"] + ) - for group, node_ids in groups.items(): - page = group_pages[group] - page.parent.mkdir(parents=True, exist_ok=True) - narrative = page.read_text(encoding="utf-8") if page.is_file() else None - chapter, linked, unresolved = _render_chapter( - group, - node_ids, - graph=graph, - statuses=statuses, - numbers=numbers, - used_by=used_by, - linker=linker, - page=page, - targets=targets, - narrative=narrative, - blueprint=blueprint, - repo_root=repo_root, - destination=destination, - node_sources=node_sources, - containers=containers, - sources_base=sources_base, + input_guard() + _require_output_parent(output_parent, "at the publication commit boundary") + + if _inspect_destination_at( + parent_descriptor, + destination.name, + destination, + allow_os_metadata=True, + ) != expected: + raise PublicationError( + ["publication inputs changed at the commit boundary"] + ) + if ( + _inspect_destination_at(stage_parent_descriptor, stage.name, stage) + != staged + or _publication_inventory_at( + stage_parent_descriptor, + stage.name, + staged.identity, + ) + != staged_inventory + ): + raise _PublicationRecoveryError( + [ + "publication stage changed; workspace retained " + f"{_workspace_recovery_location(stage.parent, output_parent)}" + ] + ) + + commit_state.attempted = True + commit = _rename_noreplace if expected.kind == "absent" else _rename_exchange + try: + commit( + stage_parent_descriptor, + stage.name, + parent_descriptor, + destination.name, + ) + except _CommitRefused: + # Nothing moved, so the stage is still ours to clean up. + commit_state.attempted = False + raise + if expected.kind == "absent": + _publication_commit_checkpoint("after-install") + else: + _publication_commit_checkpoint("after-exchange") + displaced = _inspect_destination_at( + stage_parent_descriptor, + stage.name, + stage, + allow_os_metadata=True, + ) + if displaced != expected or ( + expected.identity is not None + and expected_inventory is not None + and _publication_inventory_at( + stage_parent_descriptor, + stage.name, + expected.identity, + ) + != expected_inventory + ): + raise PublicationError( + ["the displaced publication changed during the commit operation"] + ) + + published = _inspect_destination_at( + parent_descriptor, + destination.name, + destination, ) - page.write_text(chapter, encoding="utf-8") - if narrative is None: # a milestone with no narrative page of its own - report.pages += 1 - report.nodes += len(node_ids) - report.linked += linked - report.unresolved.extend(unresolved) + if published != staged or _publication_inventory_at( + parent_descriptor, + destination.name, + staged.identity, + ) != staged_inventory: + raise PublicationError( + ["the published generation changed before its final ownership check"] + ) + os.fsync(stage_parent_descriptor) + os.fsync(parent_descriptor) + _require_output_parent(output_parent, "after publication commit") + # Cleanup only touches the private workspace; another render's commit + # will see the new generation and refuse its stale expectation. + fcntl.flock(parent_descriptor, fcntl.LOCK_UN) + except BaseException as publication_error: + if commit_state.attempted: + raise _PublicationRecoveryError( + [ + "publication commit began but its final state or durability " + f"could not be verified; output may have changed and the workspace " + f"was retained {_workspace_recovery_location(stage.parent, output_parent)}" + ] + ) from publication_error + raise + commit_state.verified = True + + +_PUBLICATION_LOCK_WAIT_SECONDS = 30.0 +_PUBLICATION_LOCK_POLL_SECONDS = 0.05 + + +def _lock_output_parent( + descriptor: int, parent: Path, expected: _DestinationState +) -> None: + """Serialize commits in one output parent, waiting for another commit to end.""" - book_pages = _book_page_order( - blueprint, - destination, - graph, - ) - # The landing page is a dashboard, not chapter one. Previous/next belongs - # to the book, so the strip starts at the contents page. - _append_book_navigation([p for p in book_pages if p != overview]) - structure = destination / STRUCTURE_PAGE - structure.write_text( - _render_structure_page( - blueprint, - graph, - statuses, - page=structure, - targets=targets, - sources_base=sources_base, - ), - encoding="utf-8", - ) - report.pages += 1 - (destination / "SUMMARY.md").write_text( - _render_summary_nav(book_pages, destination=destination, overview=overview), - encoding="utf-8", - ) + deadline = time.monotonic() + _PUBLICATION_LOCK_WAIT_SECONDS + while True: + try: + fcntl.flock(descriptor, fcntl.LOCK_EX | fcntl.LOCK_NB) + return + except BlockingIOError as error: + if time.monotonic() >= deadline: + unchanged = ( + "output was not changed" + if expected.kind == "absent" + else "previous site was preserved" + ) + raise PublicationError( + [ + f"another process kept the output parent {parent} locked for " + f"{_PUBLICATION_LOCK_WAIT_SECONDS:g} seconds; {unchanged}; retry" + ] + ) from error + time.sleep(_PUBLICATION_LOCK_POLL_SECONDS) - generated_graph_pages = graph_pages.write_graph_pages( - graph, - statuses, - destination, - node_links=lambda page: _anchored_links(targets, page), + +def _rename_noreplace( + source_parent: int, source: str, target_parent: int, target: str +) -> None: + function, flag = _rename_implementation(exchange=False) + result = function( + source_parent, + os.fsencode(source), + target_parent, + os.fsencode(target), + flag, ) - report.pages += len(generated_graph_pages) + if result == 0: + return + error = ctypes.get_errno() + if error in _REFUSED_RENAME_ERRORS: + raise _refused_commit(error, unchanged="output was not changed") + raise OSError(error, os.strerror(error), target) - for relative, contents in ( - (STYLESHEET, _stylesheet()), - (MERMAID_SCRIPT, _mermaid_script()), - (LIVE_SCRIPT, _static_asset("blueprint-live.js")), - (LOGO, _logo()), - ): - asset = destination / relative - asset.parent.mkdir(parents=True, exist_ok=True) - asset.write_text(contents, encoding="utf-8") - _write_publication_manifest( - destination, - blueprint, - graph, - linker, - coverage=coverage, - complete=True, + +def _rename_exchange( + source_parent: int, source: str, target_parent: int, target: str +) -> None: + function, flag = _rename_implementation(exchange=True) + result = function( + source_parent, + os.fsencode(source), + target_parent, + os.fsencode(target), + flag, ) - return report + if result != 0: + error = ctypes.get_errno() + if error in _REFUSED_RENAME_ERRORS: + raise _refused_commit(error, unchanged="previous site was preserved") + raise OSError(error, os.strerror(error), target) -def _prepare_destination(destination: Path, *, clean: bool) -> None: - """Create an output directory without overwriting unrelated user data.""" - if not destination.exists(): - destination.mkdir(parents=True) - return - if not destination.is_dir(): - raise PublicationError(["output path exists and is not a directory"]) - if not any(destination.iterdir()): - return +# A rename that fails changes neither name. These errors are the kernel refusing +# the request; EIO and anything unexpected stay uncertain. +_REFUSED_RENAME_ERRORS = frozenset( + { + errno.EACCES, + errno.EBUSY, + errno.EDQUOT, + errno.EEXIST, + errno.EINVAL, + errno.EISDIR, + errno.ELOOP, + errno.EMLINK, + errno.ENAMETOOLONG, + errno.ENOENT, + errno.ENOSPC, + errno.ENOSYS, + errno.ENOTDIR, + errno.ENOTEMPTY, + errno.ENOTSUP, + errno.EOPNOTSUPP, + errno.EPERM, + errno.EROFS, + errno.EXDEV, + } +) - manifest = destination / PUBLICATION_MANIFEST - publication = None - if not manifest.is_symlink() and manifest.is_file(): - try: - publication = json.loads(manifest.read_text(encoding="utf-8")) - except (OSError, UnicodeError, json.JSONDecodeError): - pass - if ( - manifest.is_symlink() - or not isinstance(publication, dict) - or publication.get("schema") != "autoform-publication/v1" - ): - raise PublicationError( - [ - "refusing to overwrite a non-Autoform output directory; " - "choose an empty directory or remove it explicitly" - ] + +class _CommitRefused(PublicationError): + """The filesystem refused the commit rename, so nothing was published.""" + + +def _refused_commit(error: int, *, unchanged: str) -> _CommitRefused: + if error in {errno.EEXIST, errno.ENOTEMPTY, errno.ENOENT}: + reason = "output directory changed during publication" + elif error == errno.EBUSY: + reason = "output directory is a mount point or in use, so it cannot be replaced" + elif error == errno.EXDEV: + reason = ( + "output directory is not on its parent's filesystem (a mount point or " + "an overlay lower layer), so it cannot be replaced" ) - if clean: - shutil.rmtree(destination) - destination.mkdir(parents=True) + else: + reason = f"the filesystem refused the publication commit: {os.strerror(error)}" + return _CommitRefused([f"{reason}; {unchanged}"]) + + +def _rename_implementation(*, exchange: bool): + try: + libc = ctypes.CDLL(None, use_errno=True) + except OSError as error: + raise PublicationError(["atomic publication is unavailable on this platform"]) from error + if hasattr(libc, "renameatx_np"): + function = libc.renameatx_np + flag = 0x00000002 if exchange else 0x00000004 + elif hasattr(libc, "renameat2"): + function = libc.renameat2 + flag = 2 if exchange else 1 + else: + raise PublicationError(["atomic publication is unavailable on this platform"]) + function.argtypes = [ + ctypes.c_int, + ctypes.c_char_p, + ctypes.c_int, + ctypes.c_char_p, + ctypes.c_uint, + ] + function.restype = ctypes.c_int + return function, flag -def _validate_publication_tree(blueprint: Path) -> None: - """Reject inputs that could leak local state through a public artifact.""" +def _validate_publication_snapshot(snapshot: TreeSnapshot) -> None: + """Reject captured entries that could leak local or non-regular state.""" + issues: list[str] = [] - if not blueprint.is_dir(): - return - for source in sorted(blueprint.rglob("*")): - relative = source.relative_to(blueprint) + paths = [ + *(relative for relative in snapshot.directories if relative), + *(relative for relative, _data in snapshot.files), + *(relative for relative, _target in snapshot.symlinks), + *(relative for relative, _mode in snapshot.special), + *(relative for relative, _kind in snapshot.omitted), + ] + symlinks = {relative for relative, _target in snapshot.symlinks} + specials = { + relative: reason + for relative, reason in snapshot.unsupported_entries() + if relative not in symlinks + } + aliased_roots: set[str] = set() + for raw_relative in sorted(paths): + relative = PurePosixPath(raw_relative) folded_parts = {part.casefold() for part in relative.parts} name = relative.name.casefold() + generated = _generated_root_alias(relative) + if generated is not None: + if relative.parts[0] not in aliased_roots: + aliased_roots.add(relative.parts[0]) + issues.append( + f"refusing blueprint root entry {relative.parts[0]}: it differs " + f"only in case from the generated {generated}, and the two would " + "collide; rename it" + ) + continue if ( folded_parts.intersection(_LOCAL_ONLY_NAMES) or name == ".env" or name.startswith(".env.") or name.endswith((".key", ".log", ".pem")) ): - issues.append(f"refusing local or sensitive publication input: {relative.as_posix()}") + issues.append( + f"refusing local or sensitive publication input: {raw_relative}" + ) continue - if _is_hidden(relative): + if any(part.startswith(".") for part in relative.parts): continue - if source.is_symlink(): - issues.append(f"refusing symlink in blueprint publication: {relative.as_posix()}") + if raw_relative in symlinks: + issues.append( + f"refusing symlink in blueprint publication: {raw_relative}" + ) + elif raw_relative in specials: + issues.append( + "refusing non-regular blueprint publication input: " + f"{raw_relative}: {specials[raw_relative]}" + ) if issues: raise PublicationError(issues) +def _load_publication_contract( + blueprint: _CapturedBlueprint, +) -> tuple[Graph, CoverageSummary]: + files = {relative.as_posix(): data for relative, data in blueprint.files.items()} + graph = load_graph_snapshot( + blueprint.root, + files, + directories=( + "" if relative == PurePosixPath(".") else relative.as_posix() + for relative in blueprint.directories + ), + ) + coverage, coverage_issues = load_coverage_snapshot(blueprint.root, files) + if coverage_issues: + raise PublicationError( + [ + f"coverage contract line {issue.line}: {issue.reason}" + if issue.line + else f"coverage contract: {issue.reason}" + for issue in coverage_issues + ] + ) + if coverage is None: + raise PublicationError(["coverage contract could not be loaded"]) + return graph, coverage + + def _is_hidden(relative: Path) -> bool: return any(part.startswith(".") for part in relative.parts) @@ -541,7 +3441,9 @@ def _sources_base(blueprint: Path, repo_root: Path, linker: SourceLinker) -> "_S if not linker.repository_url or not linker.ref: return None try: - relative = (blueprint / SOURCES_DIR).resolve().relative_to(repo_root).as_posix() + relative = _lexical_path(blueprint / SOURCES_DIR).relative_to( + _lexical_path(repo_root) + ).as_posix() except ValueError: # The vault is outside the repository being linked, so no blob URL # describes it. Better no link than one that 404s. @@ -575,40 +3477,71 @@ def _source_href(sources_base: _SourceBase, tail: tuple[str, ...]) -> str: return sources_base.href(tail) -def _published_source_files(blueprint: Path): - """Yield the regular authored inputs that contribute to the static site.""" - for source in sorted(blueprint.rglob("*")): - relative = source.relative_to(blueprint) - if _SKIPPED_DIRECTORIES.intersection(relative.parts) or _is_hidden(relative): +def _source_revision(blueprint: _CapturedBlueprint) -> str: + digest = hashlib.sha256(b"autoform-markdown-publication/v2\0") + for relative in sorted(blueprint.directories, key=lambda path: path.as_posix()): + path = Path(relative.as_posix()) + if _SKIPPED_DIRECTORIES.intersection(path.parts) or _is_hidden(path): + continue + _update_source_digest(digest, path, b"", kind=b"directory") + for relative, data in sorted( + blueprint.files.items(), key=lambda item: item[0].as_posix() + ): + path = Path(relative.as_posix()) + if _SKIPPED_DIRECTORIES.intersection(path.parts) or _is_hidden(path): continue - if relative.name in _GENERATED_FILES or not source.is_file(): + if _is_generated_path(path): continue - yield source, relative + _update_source_digest(digest, path, data) + return digest.hexdigest() -def _source_revision(blueprint: Path) -> str: - digest = hashlib.sha256(b"autoform-markdown-publication/v1\0") - for source, relative in _published_source_files(blueprint): - digest.update(relative.as_posix().encode("utf-8") + b"\0") - digest.update(source.read_bytes() + b"\0") - return digest.hexdigest() +def capture_publication_source( + blueprint_dir: str | Path, +) -> PublicationSourceSnapshot: + """Capture the exact descriptor-backed blueprint generation used by readers.""" + + try: + blueprint = Path(blueprint_dir).expanduser().resolve() + with bind_directory_tree( + blueprint, + selection=_PUBLICATION_SNAPSHOT_SELECTION, + require_descriptor=True, + ) as source_tree: + snapshot = source_tree.capture() + _validate_publication_snapshot(snapshot) + captured = _CapturedBlueprint.from_snapshot(blueprint, snapshot) + return PublicationSourceSnapshot( + root=blueprint, + files=MappingProxyType(dict(captured.files)), + directories=captured.directories, + revision=_source_revision(captured), + ) + except PublicationError as error: + raise OSError(f"blueprint source revision is not publishable: {error}") from error + except (OSError, RuntimeError, TreeSnapshotError, ValueError) as error: + raise OSError( + f"blueprint source revision could not be captured safely: {error}" + ) from error def publication_source_revision(blueprint_dir: str | Path) -> str: """Return the deterministic source hash stored in ``publication.json``.""" - return _source_revision(Path(blueprint_dir).expanduser().resolve()) + return capture_publication_source(blueprint_dir).revision def _write_publication_manifest( - destination: Path, - blueprint: Path, + plan: _PublicationPlanBuilder, graph: Graph, linker: SourceLinker, *, coverage: CoverageSummary, complete: bool, + source_revision: str, + lean_source_revision: str, ) -> None: + directories, files = plan.inventory() manifest = { "complete": complete, "coverage": { @@ -618,18 +3551,26 @@ def _write_publication_manifest( "source_path": coverage.source_path, "source_sha256": coverage.source_sha256, }, - "schema": "autoform-publication/v1", + "directories": list(directories), + "files": dict(files), + "schema": PUBLICATION_SCHEMA, "source": "blueprint/roadmap Markdown", - "source_revision": publication_source_revision(blueprint), + "source_revision": source_revision, "git_ref": linker.ref, + "lean_source_revision": lean_source_revision, "nodes": len(graph.nodes), "dependencies": graph.edge_count, "views": ["book", "progress", "project", "chapter", "focus", "full"], } - (destination / PUBLICATION_MANIFEST).write_text( - json.dumps(manifest, indent=2, sort_keys=True) + "\n", - encoding="utf-8", - ) + encoded = (json.dumps(manifest, indent=2, sort_keys=True) + "\n").encode("utf-8") + if len(encoded) > _PUBLICATION_MANIFEST_MAX_BYTES: + raise PublicationError( + [ + "generated publication manifest exceeds " + f"max_file_bytes={_PUBLICATION_MANIFEST_MAX_BYTES}" + ] + ) + plan.write_bytes(PUBLICATION_MANIFEST, encoded) def _group_nodes(graph: Graph) -> dict[str, list[str]]: @@ -664,29 +3605,34 @@ def _group_page(group: str) -> Path: ) -def _book_page_order(blueprint: Path, destination: Path, graph: Graph) -> list[Path]: +def _book_page_order( + blueprint: _CapturedBlueprint, + plan: _PublicationPlanBuilder | _PublicationFilePlan, + graph: Graph, +) -> list[Path]: """Follow authored container links to recover the book's page order.""" + destination = plan.root ordered: list[Path] = [] seen_outputs: set[Path] = set() visited_sources: set[Path] = set() containers = _containers(graph) book_sources = { - node.path.resolve() + _lexical_path(node.path) for node in graph.nodes.values() if node.id in containers or not node.formalizable } - pending = [blueprint / "README.md"] + pending = [blueprint.root / "README.md"] while pending: - source = pending.pop().resolve() + source = _lexical_path(pending.pop()) try: - relative = source.relative_to(blueprint) + relative = blueprint.relative(source) except ValueError: continue - output = (destination / relative).resolve() - if output.is_file() and output not in seen_outputs: + output = destination.joinpath(*relative.parts) + if plan.is_file(output) and output not in seen_outputs: seen_outputs.add(output) ordered.append(output) - if source in visited_sources or not source.is_file(): + if source in visited_sources or relative not in blueprint.files: continue visited_sources.add(source) linked_sources: list[Path] = [] @@ -698,9 +3644,9 @@ def collect(line: str) -> str: path = bare.partition("#")[0] if not path or urlsplit(path).scheme or path.startswith("/"): continue - candidate = (source.parent / unquote(path)).resolve() + candidate = _lexical_path(source.parent / unquote(path)) try: - candidate_relative = candidate.relative_to(blueprint) + candidate_relative = candidate.relative_to(blueprint.root) except ValueError: continue if ( @@ -714,16 +3660,20 @@ def collect(line: str) -> str: linked_sources.append(candidate) return line - _outside_fences(source.read_text(encoding="utf-8"), collect) + _outside_fences(blueprint.read_text(relative), collect) pending.extend(reversed(linked_sources)) return ordered -def _append_book_navigation(pages: list[Path]) -> None: +def _append_book_navigation( + pages: list[Path], + *, + plan: _PublicationPlanBuilder | _PublicationFilePlan, +) -> None: """Add previous/next links to the bottom of Blueprint pages, never global nav.""" if len(pages) < 2: return - titles = [_first_h1(page.read_text(encoding="utf-8")) or page.stem for page in pages] + titles = [_first_h1(plan.read_text(page)) or page.stem for page in pages] for index, page in enumerate(pages): links: list[str] = [] if index: @@ -749,9 +3699,9 @@ def _append_book_navigation(pages: list[Path]) -> None: + "".join(links) + "" ) - page.write_text( - page.read_text(encoding="utf-8").rstrip() + "\n\n" + navigation + "\n", - encoding="utf-8", + plan.write_text( + page, + plan.read_text(page).rstrip() + "\n\n" + navigation + "\n", ) @@ -809,6 +3759,7 @@ def _next_target( destination: Path, group_pages: dict[str, Path] | None = None, targets: dict[str, str] | None = None, + read_page: Callable[[Path], str], ) -> str: """Name the result a contributor could pick up right now, with the way in. @@ -849,7 +3800,7 @@ def _next_target( ) actions = [f'Dependencies'] if chapter_page is not None: - chapter_title = _first_h1(chapter_page.read_text(encoding="utf-8")) or "chapter" + chapter_title = _first_h1(read_page(chapter_page)) or "chapter" href = mermaid.relative_link(chapter_page, page, ".html") actions.insert( 0, f'{html.escape(chapter_title)}' @@ -880,7 +3831,7 @@ def _next_target( def _render_structure_page( - blueprint: Path, + blueprint: _CapturedBlueprint, graph: Graph, statuses: dict[str, status.NodeStatus], *, @@ -901,17 +3852,21 @@ def _render_structure_page( chapter is; without them every `README.md` looks like every other one. """ links = _anchored_links(targets, page, extension=".md") - by_path = {node.path.resolve(): node for node in graph.nodes.values()} + by_path = {_lexical_path(node.path): node for node in graph.nodes.values()} def keep(relative: Path) -> bool: if _SKIPPED_DIRECTORIES.intersection(relative.parts) or _is_hidden(relative): return False - if relative.name in _GENERATED_FILES: + if _is_generated_path(relative): return False return not (sources_base is not None and relative.parts[:1] == (SOURCES_DIR,)) - files = [p for p in sorted(blueprint.rglob("*.md")) if keep(p.relative_to(blueprint))] - directories = {p.relative_to(blueprint).parent for p in files} + files = [ + Path(relative.as_posix()) + for relative in sorted(blueprint.files, key=lambda path: path.as_posix()) + if relative.suffix == ".md" and keep(Path(relative.as_posix())) + ] + directories = {path.parent for path in files} directories.discard(Path(".")) for directory in list(directories): for parent in directory.parents: @@ -939,13 +3894,13 @@ def row( ) rows = [row(0, "blueprint/", "vault root", "")] - for entry in sorted(directories | {p.relative_to(blueprint) for p in files}): + for entry in sorted(directories | set(files)): depth = len(entry.parts) if entry in directories: rows.append(row(depth, f"{html.escape(entry.name)}/", "", "")) continue - source = blueprint / entry - node = by_path.get(source.resolve()) + source = blueprint.root / entry + node = by_path.get(_lexical_path(source)) name = html.escape(entry.name) if node is None: # Everything under `roadmap/` is a node or the graph refuses to @@ -967,7 +3922,7 @@ def row( ) ) - article_depths = {len(p.relative_to(blueprint).parts) - 1 for p in by_path} + article_depths = {len(p.relative_to(blueprint.root).parts) - 1 for p in by_path} warnings = [] if len(by_path) > 3 and article_depths <= {1}: warnings.append( @@ -1000,6 +3955,7 @@ def _render_summary_nav( *, destination: Path, overview: Path, + plan: _PublicationPlanBuilder | _PublicationFilePlan, ) -> str: """Write the site nav as Markdown so the Book tab holds real chapters. @@ -1012,7 +3968,7 @@ def _render_summary_nav( for page in book_pages: if page == overview: continue - title = _first_h1(page.read_text(encoding="utf-8")) or page.parent.name + title = _first_h1(plan.read_text(page)) or page.parent.name depth = len(page.relative_to(destination).parts) - 1 indent = " " * max(depth, 1) lines.append(f"{indent}- [{title}]({page.relative_to(destination).as_posix()})") @@ -1020,8 +3976,8 @@ def _render_summary_nav( # page used to be the only route to it, which made the page impossible to # simplify without stranding it. coverage = destination / "coverage/README.md" - if coverage.is_file(): - title = _first_h1(coverage.read_text(encoding="utf-8")) or "Coverage" + if plan.is_file(coverage): + title = _first_h1(plan.read_text(coverage)) or "Coverage" lines.append(f" - [{title}](coverage/README.md)") lines.extend( [ @@ -1044,6 +4000,7 @@ def _render_landing_page( page: Path, destination: Path, targets: dict[str, str] | None = None, + read_page: Callable[[Path], str], ) -> str: """The landing page: what this is, how far it has got, and what is next. @@ -1087,6 +4044,7 @@ def _render_landing_page( destination=destination, group_pages=group_pages, targets=targets, + read_page=read_page, ), ] breakdown = mermaid.render_legend(statuses) @@ -1363,7 +4321,7 @@ def _anchored_links( and extension can share *hrefs*, so each target page is linked once across all of them. """ - resolved_page = page.resolve() + resolved_page = _lexical_path(page) # Many nodes share a chapter page, and resolving a path walks the disk, so # each target page is linked once. The current page is cached as "" and # links as a bare fragment. @@ -1373,7 +4331,7 @@ def _anchored_links( for node_id, (target, anchor) in targets.items(): href = hrefs.get(target) if href is None: - if target.resolve() == resolved_page: + if _lexical_path(target) == resolved_page: href = "" else: href = mermaid.relative_link(target, page, extension) @@ -1429,7 +4387,7 @@ def moved_target(raw: str) -> str | None: path, separator, fragment = bare.partition("#") if not path or urlsplit(path).scheme or path.startswith("/"): return None - candidate = (source_dir / unquote(path)).resolve() + candidate = _lexical_path(source_dir / unquote(path)) node_id = node_sources.get(candidate) if node_id is not None: href = _anchored_links({node_id: targets[node_id]}, page, extension=".md", hrefs=page_hrefs)[node_id] @@ -1472,6 +4430,53 @@ def _is_within(path: Path, directory: Path) -> bool: return True +def _lexical_path(path: Path) -> Path: + return Path(os.path.normpath(os.fspath(path))) + + +def _publication_paths_overlap(first: Path, second: Path) -> bool: + """Recognize lexical and filesystem aliases before creating render state.""" + + return ( + _is_within(first, second) + or _is_within(second, first) + or _path_reaches_directory_identity(first, second) + or _path_reaches_directory_identity(second, first) + ) + + +def _path_reaches_directory_identity(path: Path, directory: Path) -> bool: + try: + target = directory.stat() + except FileNotFoundError: + return False + except OSError as error: + raise PublicationError(["could not verify publication path separation"]) from error + if not stat.S_ISDIR(target.st_mode): + return False + target_identity = (target.st_dev, target.st_ino) + cursor = path + while True: + try: + metadata = cursor.stat() + except FileNotFoundError: + pass + except OSError as error: + raise PublicationError( + ["could not verify publication path separation"] + ) from error + else: + if stat.S_ISDIR(metadata.st_mode) and ( + metadata.st_dev, + metadata.st_ino, + ) == target_identity: + return True + parent = cursor.parent + if parent == cursor: + return False + cursor = parent + + def _outside_fences(text: str, transform) -> str: """Apply *transform* to every line that is not inside a code fence.""" fence: tuple[str, int] | None = None @@ -1531,6 +4536,8 @@ def _render_chapter( node_sources: dict[Path, str], containers: frozenset[str], sources_base: "_SourceBase | None" = None, + source_blueprint: Path, + read_source: Callable[[Path], str], ) -> tuple[str, int, list[str]]: """Render one narrative article with statements at its authored link slots.""" links = _anchored_links(targets, page) @@ -1554,6 +4561,8 @@ def _render_chapter( node_sources=node_sources, targets=targets, sources_base=sources_base, + source_blueprint=source_blueprint, + read_source=read_source, ) environments[node_id] = environment linked += node_linked @@ -1618,7 +4627,7 @@ def _place_environments( elif not path or urlsplit(path).scheme or path.startswith("/"): node_id = None else: - node_id = node_sources.get((source_dir / unquote(path)).resolve()) + node_id = node_sources.get(_lexical_path(source_dir / unquote(path))) if node_id is None or node_id not in environments or node_id in placed: output.append(line) continue @@ -1644,10 +4653,12 @@ def _render_environment( node_sources: dict[Path, str], targets: dict[str, tuple[Path, str]], sources_base: "_SourceBase | None" = None, + source_blueprint: Path, + read_source: Callable[[Path], str], ) -> tuple[str, int, list[str]]: node_status = statuses[node.id] caption, _, number = numbers[node.id].rpartition(" ") - statement, remainder = _split_body(node.path.read_text(encoding="utf-8")) + statement, remainder = _split_body(read_source(node.path)) # The body is leaving its own directory for the chapter page, so its # relative links have to be recomputed from the chapter's location. statement, remainder = ( @@ -1666,7 +4677,13 @@ def _render_environment( code_links, implementation_rows, linked, unresolved = _lean_presentation(node, linker) context_link = _graph_context_link(node, page=page, destination=destination) - source_link = _vault_source_link(node, repo_root=repo_root, linker=linker) + source_link = _vault_source_link( + node, + blueprint=blueprint, + source_blueprint=source_blueprint, + repo_root=repo_root, + linker=linker, + ) meta_rows = implementation_rows if node_status.key == "conditional": # A conditional proof must never read as finished, so the open @@ -1767,7 +4784,14 @@ def _code_icon() -> str: ) -def _vault_source_link(node: Node, *, repo_root: Path, linker) -> str: +def _vault_source_link( + node: Node, + *, + blueprint: Path, + source_blueprint: Path, + repo_root: Path, + linker, +) -> str: """Link a statement to the Markdown article it was authored in. The graph view and the published statement are both derived. This is the @@ -1776,10 +4800,11 @@ def _vault_source_link(node: Node, *, repo_root: Path, linker) -> str: if not linker.repository_url or not linker.ref: return "" try: - relative = node.path.resolve().relative_to(repo_root).as_posix() + article = source_blueprint / _lexical_path(node.path).relative_to(blueprint) + relative = article.relative_to(repo_root).as_posix() except ValueError: return "" - href = f"{linker.repository_url}/blob/{linker.ref}/{relative}" + href = f"{linker.repository_url}/blob/{linker.ref}/{quote(relative, safe='/')}" label = html.escape(f"Edit the Markdown source for {node.title}", quote=True) icon = ( '
0%
' in overview + assert '
0
' in overview + + +def test_a_directory_link_uses_tree_even_when_the_repo_url_says_blob() -> None: + """Deriving the directory URL by replacing the first `/blob/` rewrote the + repository's own path when that happened to contain one.""" + from autoform_cli.render import _SourceBase + + base = _SourceBase("https://git.example/blob/x/repo", "abc", "blueprint/sources") + + assert base.href(("paper.md",)) == ( + "https://git.example/blob/x/repo/blob/abc/blueprint/sources/paper.md" + ) + assert base.href(()) == "https://git.example/blob/x/repo/tree/abc/blueprint/sources" + + +def _count_relative_links(monkeypatch: pytest.MonkeyPatch) -> list[Path]: + """Record the target of every `mermaid.relative_link` call.""" + from autoform_cli import mermaid + + calls: list[Path] = [] + relative_link = mermaid.relative_link + + def counting(target: Path, output: Path, link_extension: str) -> str: + calls.append(target) + return relative_link(target, output, link_extension) + + monkeypatch.setattr(mermaid, "relative_link", counting) + return calls + + +def test_node_links_resolve_each_target_page_once(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """Building a link per node resolved paths on disk once per node on every + page, which made render time grow with the square of the node count.""" + from autoform_cli.render import _anchored_links + + chapter, other, page = tmp_path / "a.md", tmp_path / "b" / "README.md", tmp_path / "page.md" + targets = {f"a/{index}": (chapter, f"n{index}") for index in range(20)} + targets["b"] = (other, "") + targets["here"] = (page, "self") + calls = _count_relative_links(monkeypatch) + resolved: list[Path] = [] + resolve = Path.resolve + + def counting_resolve(self: Path, strict: bool = False) -> Path: + resolved.append(self) + return resolve(self, strict) + + monkeypatch.setattr(Path, "resolve", counting_resolve) + + links = _anchored_links(targets, page) + + assert sorted(calls) == [chapter, other] + # Once to compare it with the current page, once inside relative_link. + assert resolved.count(chapter) <= 2 + assert links["a/7"] == "a.html#n7" + assert links["b"] == "b/index.html" + assert links["here"] == "#self" + + +def test_rewriting_a_page_links_only_the_nodes_it_names(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """A page uses the few node links it contains, so rewriting it should not + build a link for every node in the graph.""" + from autoform_cli.render import _rewrite_links + + blueprint, destination = tmp_path / "blueprint", tmp_path / "out" + source_dir = blueprint / "roadmap" + source_dir.mkdir(parents=True) + chapter, other, page = destination / "a.md", destination / "b.md", destination / "page.md" + targets = {"a/x": (chapter, "x"), "a/z": (chapter, "z"), "b/y": (other, "y")} + node_sources = { + (source_dir / "x.md").resolve(): "a/x", + (source_dir / "z.md").resolve(): "a/z", + (source_dir / "y.md").resolve(): "b/y", + } + calls = _count_relative_links(monkeypatch) + + def rewrite(text: str) -> str: + return _rewrite_links( + text, + source_dir=source_dir, + page=page, + blueprint=blueprint, + destination=destination, + node_sources=node_sources, + targets=targets, + ) + + assert rewrite("Plain prose.\n") == "Plain prose.\n" + assert calls == [] + assert rewrite("See [X](x.md).\n") == "See [X](a.md#x).\n" + assert calls == [chapter] + calls.clear() + # Two nodes on the same chapter page cost one link to it, not one each. + assert rewrite("[X](x.md), [Z](z.md)\n") == "[X](a.md#x), [Z](a.md#z)\n" + assert calls == [chapter] + + +def test_a_focus_page_asks_for_its_node_links_once(tmp_path: Path) -> None: + """Each request builds a link for every node, so asking twice per focus + page doubled the cost of the largest group of generated pages.""" + from autoform_cli.graph_pages import focus_page_path, write_graph_pages + + project = _project(tmp_path) + graph = load_graph(project / "blueprint") + destination = tmp_path / "out" + requested: list[Path] = [] + + def node_links(page: Path) -> dict[str, str]: + requested.append(page) + return {node_id: f"roadmap.html#{node_id}" for node_id in graph.nodes} + + write_graph_pages(graph, derive(graph), destination, node_links=node_links) + + for node_id in ("base", "top"): + page = focus_page_path(destination, node_id) + assert requested.count(page) == 1 + assert f"roadmap.html#{node_id}" in page.read_text(encoding="utf-8") + + +def test_a_node_that_is_the_current_page_links_as_a_bare_fragment(tmp_path: Path) -> None: + """A container article has no anchor of its own, and an empty href would + drop the link, so its own page links to the top of itself.""" + from autoform_cli.render import _anchored_links + + page = tmp_path / "chapter" / "README.md" + + assert _anchored_links({"chapter": (page, ""), "chapter/x": (page, "x")}, page) == { + "chapter": "#", + "chapter/x": "#x", + } + + +def _conditional_project(tmp_path: Path, policy: str) -> Path: + """`_project` with Top proved from an open statement, under the given policy.""" + project = _project(tmp_path) + roadmap = project / "blueprint" / "roadmap" + (roadmap / "README.md").write_text( + f"---\nopen_statements: {policy}\n---\n\n# Roadmap\n\n" + "## Definitions\n\n- [Base](base.md)\n\n" + "## Results\n\n- [Open](open.md)\n- [Top](top.md)\n", + encoding="utf-8", + ) + (project / "Project" / "Open.lean").write_text( + "namespace Project\n\ntheorem openStatement : True := sorry\n\nend Project\n", encoding="utf-8" + ) + (roadmap / "open.md").write_text( + "---\ndeclaration: theorem\nstatement: formalized\nlean: Project.openStatement\n---\n\n" + "# Open\n\nA statement whose proof is still sorry.\n\n## Depends on\n\n- [Base](base.md)\n", + encoding="utf-8", + ) + top = roadmap / "top.md" + top.write_text( + top.read_text(encoding="utf-8") + "\n## Proof depends on\n\n- [Open](open.md)\n", encoding="utf-8" + ) + return project + + +def test_a_conditional_proof_names_the_open_statements_it_assumes( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + # The Lean row is the fallback for a declaration without a source link, and + # on CI the Actions environment would supply the coordinates for one. + for variable in ("GITHUB_REPOSITORY", "GITHUB_SERVER_URL", "GITHUB_SHA"): + monkeypatch.delenv(variable, raising=False) + project = _conditional_project(tmp_path, "allowed") + + render_site(project / "blueprint", tmp_path / "out", lean_root=project, repository_url="", ref="") + page = (tmp_path / "out/roadmap/README.md").read_text(encoding="utf-8") + top = page[page.index('id="top"'):] + + assert '
conditionally proved' in top + assert ( + 'Assumes' + 'Theorem 1 (Open)' + " (open statements without a recorded Lean proof)" + ) in top + # Only the conditional proof carries the row; the open statement assumes nothing. + assert page.count('Assumes') == 1 + # The row follows the implementation row and precedes Discussion. + meta = top[top.index('
'):top.index('
')] + assert re.findall(r'([^<]+)', meta) == ["Lean", "Assumes", "Discussion"] + + css = (tmp_path / "out/stylesheets/blueprint.css").read_text(encoding="utf-8") + assert ".bp-ref-conditional::before, .bp-swatch-conditional { background: #E9DFFC; border-color: #6B3FCF; }" in css + assert "[data-md-color-scheme=slate] .bp-ref-conditional::before" in css + assert ".bp-conditional .bp-mark { color: #6B3FCF; }" in css + + +def test_the_strict_policy_shows_the_same_proof_as_proved_with_no_assumptions(tmp_path: Path) -> None: + project = _conditional_project(tmp_path, "forbidden") + + render_site(project / "blueprint", tmp_path / "out", lean_root=project) + page = (tmp_path / "out/roadmap/README.md").read_text(encoding="utf-8") + + assert '
Assumes' not in page + + +@pytest.mark.parametrize( + ("dropped", "why"), + [ + ("proof: formalized\n", "Its prerequisites are ready, so the proof can be written now."), + ( + "statement: formalized\nproof: formalized\n", + "Its prerequisites are ready, so the statement can be written down.", + ), + ], +) +def test_next_up_explains_readiness_without_naming_a_policy(tmp_path: Path, dropped: str, why: str) -> None: project = _project(tmp_path) - sources = project / "blueprint" / "sources" - sources.mkdir() - (sources / "paper note.md").write_text("---\n---\n\n# Paper\n", encoding="utf-8") - roadmap = project / "blueprint" / "roadmap" / "README.md" - roadmap.write_text( - roadmap.read_text(encoding="utf-8") - + '\nGrounded in [Paper][paper].\n\n[paper]: <../sources/paper note.md> "Source note"\n', + top = project / "blueprint" / "roadmap" / "top.md" + top.write_text(top.read_text(encoding="utf-8").replace(dropped, ""), encoding="utf-8") + + render_site(project / "blueprint", tmp_path / "out", lean_root=project) + landing = (tmp_path / "out/README.md").read_text(encoding="utf-8") + + assert '
' in landing + assert f'
{why}
' in landing + + +def test_nested_authored_page_named_like_generated_output_is_published( + tmp_path: Path, +) -> None: + project = _project(tmp_path) + (project / "blueprint/roadmap/dependencies.md").write_text( + "---\ndeclaration: theorem\n---\n\n# Authored dependencies\n\nA theorem.\n", encoding="utf-8", ) - out = tmp_path / "out" - render_site( + report = render_site(project / "blueprint", tmp_path / "out", lean_root=project) + + assert report.nodes == 3 + chapter = (tmp_path / "out/roadmap/README.md").read_text(encoding="utf-8") + assert "Authored dependencies" in chapter + + +@pytest.mark.parametrize( + ("alias", "generated"), + ( + ("Dependencies.md", "dependencies.md"), + ("Structure.md", "structure.md"), + ("Publication.json", "publication.json"), + ("Summary.md", "SUMMARY.md"), + ("Assets/figure.svg", "assets"), + ("Stylesheets/extra.css", "stylesheets"), + ), +) +def test_root_case_alias_of_generated_output_is_rejected( + tmp_path: Path, + alias: str, + generated: str, +) -> None: + project = _project(tmp_path) + authored = project / "blueprint" / alias + authored.parent.mkdir(parents=True, exist_ok=True) + authored.write_text("# Authored page\n", encoding="utf-8") + output = tmp_path / "out" + + with pytest.raises( + PublicationError, + match=f"differs only in case from the generated {re.escape(generated)}", + ) as error: + render_site(project / "blueprint", output, lean_root=project) + + assert len(error.value.issues) == 1 + assert authored.read_text(encoding="utf-8") == "# Authored page\n" + assert not output.exists() + assert not list(tmp_path.glob(f"{render_module._PUBLICATION_STAGE_PREFIX}*")) + + +@pytest.mark.parametrize("name", ("Progress.md", "Graph.html", "Dependencies.html")) +def test_root_names_render_no_longer_generates_are_published( + tmp_path: Path, + name: str, +) -> None: + project = _project(tmp_path) + (project / "blueprint" / name).write_text("# Authored page\n", encoding="utf-8") + output = tmp_path / "out" + + render_site(project / "blueprint", output, lean_root=project) + + assert name in json.loads((output / PUBLICATION_MANIFEST).read_text(encoding="utf-8"))["files"] + + +def test_output_must_have_an_ordinary_final_component(tmp_path: Path) -> None: + project = _project(tmp_path) + + with pytest.raises(PublicationError, match="ordinary directory"): + render_site(project / "blueprint", tmp_path / "child" / "..", lean_root=project) + + assert not (tmp_path / "child").exists() + + +def test_output_parent_binding_uses_one_retained_final_descriptor( + tmp_path: Path, +) -> None: + parent = tmp_path / "new" / "nested" / "parent" + + binding = render_module._open_or_create_output_parent(parent) + try: + metadata = os.fstat(binding.descriptor) + assert binding.path == parent.absolute() + assert binding.identity == (metadata.st_dev, metadata.st_ino) + assert not hasattr(binding, "descriptors") + assert not hasattr(binding, "identities") + binding.verify() + finally: + binding.close() + + +def test_render_refuses_portable_blueprint_capture( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + monkeypatch.setattr( + tree_snapshot_module, + "_DESCRIPTOR_CAPTURE_SUPPORTED", + False, + ) + + with pytest.raises(PublicationError, match="requires safe directory traversal"): + render_site(project / "blueprint", tmp_path / "out", lean_root=project) + + assert not (tmp_path / "out").exists() + + +def test_render_rejects_a_case_alias_destination_inside_the_blueprint( + tmp_path: Path, +) -> None: + project = _project(tmp_path) + blueprint = project / "blueprint" + alias = project / "BLUEPRINT" + if not alias.exists(): + pytest.skip("filesystem is case-sensitive") + + output = alias / "site" + with pytest.raises(PublicationError, match="must be disjoint"): + render_site(blueprint, output, lean_root=project) + + assert not output.exists() + assert not any( + path.name.startswith(render_module._PUBLICATION_STAGE_PREFIX) + for path in blueprint.iterdir() + ) + + +def test_render_reports_visible_special_file_before_publication(tmp_path: Path) -> None: + if not hasattr(os, "mkfifo"): + pytest.skip("FIFOs are unavailable") + project = _project(tmp_path) + fifo = project / "blueprint/roadmap/trap.md" + os.mkfifo(fifo) + output = tmp_path / "out" + + with pytest.raises(PublicationError, match=r"trap\.md: named pipe"): + render_site(project / "blueprint", output, lean_root=project) + + assert not output.exists() + + +def test_render_cleans_up_a_workspace_containing_a_near_name_max_file( + tmp_path: Path, +) -> None: + project = _project(tmp_path) + filename = "a" * 240 + ".txt" + (project / "blueprint" / filename).write_text("large name\n", encoding="utf-8") + output = tmp_path / "out" + + report = render_site(project / "blueprint", output, lean_root=project) + + assert report.warnings == [] + assert (output / filename).read_text(encoding="utf-8") == "large name\n" + assert not list(tmp_path.glob(f"{render_module._PUBLICATION_STAGE_PREFIX}*")) + + +def test_blueprint_reselection_during_capture_is_rejected( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + other_project = _project(tmp_path / "other") + blueprint = project / "blueprint" + replacement = other_project / "blueprint" + retained = tmp_path / "retained-blueprint" + output = tmp_path / "out" + swapped = False + + def swap_after_root_list(event: str, relative: str) -> None: + nonlocal swapped + if not swapped and event == "after-directory-list" and not relative: + blueprint.rename(retained) + replacement.rename(blueprint) + swapped = True + + monkeypatch.setattr( + tree_snapshot_module, + "_tree_snapshot_checkpoint", + swap_after_root_list, + ) + try: + with pytest.raises(PublicationError, match="blueprint changed"): + render_site(blueprint, output, lean_root=project) + finally: + if swapped: + blueprint.rename(replacement) + retained.rename(blueprint) + + assert not output.exists() + + +def test_blueprint_a_b_a_change_during_render_is_rejected( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + article = project / "blueprint/roadmap/top.md" + stable = article.read_text(encoding="utf-8") + original = render_module._build_publication_plan + + def mutate_and_restore(*args, **kwargs): + report = original(*args, **kwargs) + article.write_text(stable + "\nTransient.\n", encoding="utf-8") + article.write_text(stable, encoding="utf-8") + return report + + monkeypatch.setattr(render_module, "_build_publication_plan", mutate_and_restore) + + with pytest.raises(PublicationError, match="blueprint changed"): + render_site(project / "blueprint", output, lean_root=project) + + assert not output.exists() + + +def test_git_remote_a_b_a_change_cannot_change_published_links( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + for variable in ("GITHUB_REPOSITORY", "GITHUB_SERVER_URL"): + monkeypatch.delenv(variable, raising=False) + project = _project(tmp_path) + ref = _commit_project(project) + output = tmp_path / "out" + subprocess.run( + ["git", "config", "remote.origin.url", "https://github.com/correct/source.git"], + cwd=project, + check=True, + ) + original = render_module.build_linker + + def expose_transient_remote(*args, **kwargs): + assert kwargs["detect_missing"] is False + subprocess.run( + ["git", "config", "remote.origin.url", "https://github.com/wrong/source.git"], + cwd=project, + check=True, + ) + try: + return original(*args, **kwargs) + finally: + subprocess.run( + [ + "git", + "config", + "remote.origin.url", + "https://github.com/correct/source.git", + ], + cwd=project, + check=True, + ) + + monkeypatch.setattr(render_module, "build_linker", expose_transient_remote) + render_site(project / "blueprint", output, lean_root=project, ref=ref) + + page = (output / "roadmap/README.md").read_text(encoding="utf-8") + assert "github.com/correct/source" in page + assert "github.com/wrong/source" not in page + + +def test_transient_auto_detected_remote_cannot_label_captured_sources( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + for variable in ("GITHUB_REPOSITORY", "GITHUB_SERVER_URL"): + monkeypatch.delenv(variable, raising=False) + project = _project(tmp_path) + ref = _commit_project(project) + subprocess.run( + ["git", "config", "remote.origin.url", "https://github.com/correct/source.git"], + cwd=project, + check=True, + ) + original = render_module.detect_repository_url + first = True + + def transient_remote(root: Path) -> str | None: + nonlocal first + if not first: + return original(root) + first = False + subprocess.run( + ["git", "config", "remote.origin.url", "https://github.com/wrong/source.git"], + cwd=project, + check=True, + ) + try: + return original(root) + finally: + subprocess.run( + [ + "git", + "config", + "remote.origin.url", + "https://github.com/correct/source.git", + ], + cwd=project, + check=True, + ) + + monkeypatch.setattr(render_module, "detect_repository_url", transient_remote) + report = render_site( project / "blueprint", - out, + tmp_path / "out", lean_root=project, - repository_url="https://github.com/owner/repo", - ref="cafe1234", + ref=ref, + ) + + page = (tmp_path / "out/roadmap/README.md").read_text(encoding="utf-8") + assert "github.com/correct/source" not in page + assert "github.com/wrong/source" not in page + assert report.warnings + + +def test_blueprint_capture_enforces_file_limit( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + existing = [ + path.stat().st_size + for path in (project / "blueprint").rglob("*") + if path.is_file() + ] + maximum = max(existing) + (project / "blueprint/oversized.bin").write_bytes(b"x" * (maximum + 1)) + monkeypatch.setattr( + render_module, + "_PUBLICATION_SNAPSHOT_SELECTION", + TreeSelection( + include=render_module._publication_snapshot_includes, + descend=render_module._publication_snapshot_descends, + limits=TreeCaptureLimits( + max_entries=10_000, + max_depth=128, + max_file_bytes=maximum, + max_total_bytes=10 * 1024 * 1024, + ), + ), + ) + + with pytest.raises(PublicationError, match="max_file_bytes"): + render_site(project / "blueprint", tmp_path / "out", lean_root=project) + + assert not (tmp_path / "out").exists() + + +def test_lean_capture_enforces_file_limit( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + monkeypatch.setattr( + render_module, + "_LEAN_CAPTURE_LIMITS", + TreeCaptureLimits( + max_depth=128, + max_file_bytes=8, + max_total_bytes=1024, + ), + ) + + with pytest.raises(PublicationError, match="Lean source.*max_file_bytes"): + render_site(project / "blueprint", tmp_path / "out", lean_root=project) + + assert not (tmp_path / "out").exists() + + +def test_untracked_trees_in_the_lean_root_do_not_spend_the_publication_budget( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + environment = project / "venv" / "lib" + environment.mkdir(parents=True) + for index in range(50): + (environment / f"module{index}.py").write_text("", encoding="utf-8") + # A non-dot venv/ or node_modules/ can hold far more than the publication's + # entry cap; only the published tree is bounded by it. + monkeypatch.setattr( + render_module, + "_PUBLICATION_CAPTURE_LIMITS", + TreeCaptureLimits(max_entries=20, max_depth=128), + ) + + report = render_site(project / "blueprint", tmp_path / "out", lean_root=project) + + assert report.linked == 2 + + +@pytest.mark.skipif( + not hasattr(os, "geteuid") or os.geteuid() == 0, + reason="root ignores directory read permission", +) +def test_render_accepts_an_ancestor_that_grants_search_but_not_read( + tmp_path: Path, +) -> None: + home = tmp_path / "home" + project = _project(home / "alice") + output = project / "site-src" + home.chmod(0o311) + try: + render_site(project / "blueprint", output, lean_root=project) + render_site(project / "blueprint", output, lean_root=project) + finally: + home.chmod(0o755) + + assert json.loads((output / PUBLICATION_MANIFEST).read_text(encoding="utf-8"))["complete"] + assert not list(project.glob(f"{render_module._PUBLICATION_STAGE_PREFIX}*")) + + +def test_manifest_read_is_bounded( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + render_site(project / "blueprint", output, lean_root=project) + before = (output / PUBLICATION_MANIFEST).read_bytes() + monkeypatch.setattr(render_module, "_PUBLICATION_MANIFEST_MAX_BYTES", 8) + + with pytest.raises(PublicationError, match="max_file_bytes=8"): + render_site(project / "blueprint", output, lean_root=project) + + assert (output / PUBLICATION_MANIFEST).read_bytes() == before + + +def test_generated_manifest_is_bounded_before_staging( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + monkeypatch.setattr(render_module, "_PUBLICATION_MANIFEST_MAX_BYTES", 8) + + with pytest.raises(PublicationError, match="generated publication manifest.*max_file_bytes=8"): + render_site(project / "blueprint", output, lean_root=project) + + assert not output.exists() + assert not list(tmp_path.glob(".autoform-publication-*")) + + +def test_inventory_enumeration_is_bounded( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + root = tmp_path / "tree" + root.mkdir() + for name in ("a", "b", "c"): + (root / name).write_text(name, encoding="utf-8") + monkeypatch.setattr(render_module, "_PUBLICATION_MAX_ENTRIES", 2) + + with pytest.raises(PublicationError, match="max_entries=2"): + render_module._cleanup_inventory(root) + + +@pytest.mark.parametrize( + ("target", "retained_name"), + (("workspace", "."), ("site", "site")), +) +def test_post_commit_nested_injection_is_retained_instead_of_deleted( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + target: str, + retained_name: str, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + if target == "site": + render_site(project / "blueprint", output, lean_root=project) + victim = tmp_path / "victim" + victim.mkdir() + (victim / "PRECIOUS").write_text("keep\n", encoding="utf-8") + original = render_module._publish_staged_site + + def inject_after_publish(*args, **kwargs): + original(*args, **kwargs) + workspace = Path(args[0]).parent + root = workspace if target == "workspace" else workspace / target + victim.rename(root / "injected-victim") + + monkeypatch.setattr(render_module, "_publish_staged_site", inject_after_publish) + + report = render_site(project / "blueprint", output, lean_root=project) + + assert (output / PUBLICATION_MANIFEST).is_file() + assert len(report.warnings) == 1 + workspace = Path(report.warnings[0].rsplit(" at ", 1)[1]) + assert (workspace / retained_name / "injected-victim/PRECIOUS").read_text( + encoding="utf-8" + ) == "keep\n" + + +def test_post_commit_displaced_generation_disappearance_is_not_reported_as_cleaned( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + render_site(project / "blueprint", output, lean_root=project) + original = render_module._publish_staged_site + + def remove_displaced_after_publish(*args, **kwargs): + original(*args, **kwargs) + shutil.rmtree(Path(args[0]).parent / "site") + + monkeypatch.setattr( + render_module, + "_publish_staged_site", + remove_displaced_after_publish, + ) + + report = render_site(project / "blueprint", output, lean_root=project) + + assert (output / PUBLICATION_MANIFEST).is_file() + assert len(report.warnings) == 1 + workspace = Path(report.warnings[0].rsplit(" at ", 1)[1]) + assert workspace.is_dir() + + +def test_cleanup_claim_retains_a_replacement_injected_after_read( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + render_site(project / "blueprint", output, lean_root=project) + victim = tmp_path / "victim" + victim.write_text("PRECIOUS\n", encoding="utf-8") + original = render_module._read_regular_file_at + injected = False + + def inject_after_read( + parent_descriptor, + name, + display_path, + *, + max_bytes, + ignore_close_errors=False, + ): + nonlocal injected + data = original( + parent_descriptor, + name, + display_path, + max_bytes=max_bytes, + ignore_close_errors=ignore_close_errors, + ) + if ".autoform-cleanup-" in name and not injected: + injected = True + displaced = f"{name}.displaced" + os.rename( + name, + displaced, + src_dir_fd=parent_descriptor, + dst_dir_fd=parent_descriptor, + ) + os.rename(victim, name, dst_dir_fd=parent_descriptor) + return data + + monkeypatch.setattr(render_module, "_read_regular_file_at", inject_after_read) + + report = render_site(project / "blueprint", output, lean_root=project) + + assert injected + assert len(report.warnings) == 1 + workspace = Path(report.warnings[0].rsplit(" at ", 1)[1]) + retained = [ + path + for path in workspace.rglob("*") + if path.is_file() and path.read_text(encoding="utf-8") == "PRECIOUS\n" + ] + assert len(retained) == 1 + + +def test_nested_cleanup_disappearance_is_not_reported_as_success( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + workspace = tmp_path / ".autoform-publication-test" + child = workspace / "source" + child.mkdir(parents=True) + (child / "a").write_text("a", encoding="utf-8") + disappearing = child / "b" + disappearing.write_text("b", encoding="utf-8") + workspace_identity = render_module._directory_path_identity(workspace) + child_identity = render_module._directory_path_identity(child) + inventory = render_module._cleanup_inventory(child) + original = render_module._cleanup_rename_noreplace + + def remove_sibling(descriptor, source, target): + original(descriptor, source, target) + if source == "a" and disappearing.exists(): + os.unlink("b", dir_fd=descriptor) + + monkeypatch.setattr(render_module, "_cleanup_rename_noreplace", remove_sibling) + + assert not render_module._remove_owned_workspace( + workspace, + workspace_identity, + expected_children={"source": {child_identity: inventory}}, + ) + assert workspace.exists() + + +def test_verified_replacement_cleanup_never_deletes_a_swapped_new_generation( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + render_site(project / "blueprint", output, lean_root=project) + previous_manifest = (output / PUBLICATION_MANIFEST).read_bytes() + article = project / "blueprint/roadmap/top.md" + article.write_text( + article.read_text(encoding="utf-8") + "\nNew generation.\n", + encoding="utf-8", ) - - chapter = (out / "roadmap/README.md").read_text(encoding="utf-8") - assert ( - '[paper]: https://github.com/owner/repo/blob/cafe1234/blueprint/sources/paper%20note.md "Source note"' - in chapter + original = render_module._publish_staged_site + + def restore_previous_after_verified_commit(stage, destination, *args, **kwargs): + original(stage, destination, *args, **kwargs) + published = tmp_path / "published-generation" + destination.rename(published) + stage.rename(destination) + published.rename(stage) + + monkeypatch.setattr( + render_module, + "_publish_staged_site", + restore_previous_after_verified_commit, ) - assert "<../sources/paper note.md>" not in chapter + report = render_site(project / "blueprint", output, lean_root=project) -def test_a_fresh_vault_reports_no_work_rather_than_one_ready_item(tmp_path: Path) -> None: - """The roadmap landing page is not a formalization target. + assert (output / PUBLICATION_MANIFEST).read_bytes() == previous_manifest + assert len(report.warnings) == 1 + workspace = Path(report.warnings[0].rsplit(" at ", 1)[1]) + assert (workspace / "site/publication.json").read_bytes() != previous_manifest - Counting every childless article made a freshly scaffolded vault claim - "0 of 1 targets complete, 1 ready now", so the site described work before any - had been planned. - """ - from autoform_cli.scaffold import scaffold_project - project = tmp_path / "project" - scaffold_project(project, title="Empty") - out = tmp_path / "out" +def test_verified_replacement_reports_a_whole_workspace_disappearance( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + render_site(project / "blueprint", output, lean_root=project) + previous_manifest = (output / PUBLICATION_MANIFEST).read_bytes() + article = project / "blueprint/roadmap/top.md" + article.write_text( + article.read_text(encoding="utf-8") + "\nNew generation.\n", + encoding="utf-8", + ) + moved_workspace = tmp_path / "moved-publication-workspace" + original = render_module._publish_staged_site - render_site(project / "blueprint", out) + def move_workspace_after_verified_commit(stage, destination, *args, **kwargs): + original(stage, destination, *args, **kwargs) + stage.parent.rename(moved_workspace) - overview = (out / "README.md").read_text(encoding="utf-8") - assert "0 of 0 targets complete" in overview - assert "items settled" not in overview - assert '
0%
' in overview - assert '
0
' in overview + monkeypatch.setattr( + render_module, + "_publish_staged_site", + move_workspace_after_verified_commit, + ) + report = render_site(project / "blueprint", output, lean_root=project) -def test_a_directory_link_uses_tree_even_when_the_repo_url_says_blob() -> None: - """Deriving the directory URL by replacing the first `/blob/` rewrote the - repository's own path when that happened to contain one.""" - from autoform_cli.render import _SourceBase + assert (output / PUBLICATION_MANIFEST).read_bytes() != previous_manifest + assert len(report.warnings) == 1 + assert "cleanup was refused" in report.warnings[0] + assert (moved_workspace / "site/publication.json").read_bytes() == previous_manifest - base = _SourceBase("https://git.example/blob/x/repo", "abc", "blueprint/sources") - assert base.href(("paper.md",)) == ( - "https://git.example/blob/x/repo/blob/abc/blueprint/sources/paper.md" +def test_precommit_cleanup_never_deletes_a_moved_prior_generation( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + render_site(project / "blueprint", output, lean_root=project) + previous_manifest = (output / PUBLICATION_MANIFEST).read_bytes() + article = project / "blueprint/roadmap/top.md" + article.write_text( + article.read_text(encoding="utf-8") + "\nNew generation.\n", + encoding="utf-8", ) - assert base.href(()) == "https://git.example/blob/x/repo/tree/abc/blueprint/sources" + intended_stage = tmp_path / "intended-stage" + def move_prior_into_workspace(stage, destination, *args, **kwargs): + stage.rename(intended_stage) + destination.rename(stage) + raise RuntimeError("injected before commit") -def _count_relative_links(monkeypatch: pytest.MonkeyPatch) -> list[Path]: - """Record the target of every `mermaid.relative_link` call.""" - from autoform_cli import mermaid + monkeypatch.setattr( + render_module, + "_publish_staged_site", + move_prior_into_workspace, + ) - calls: list[Path] = [] - relative_link = mermaid.relative_link + with pytest.raises(PublicationError, match="cleanup was refused") as error: + render_site(project / "blueprint", output, lean_root=project) - def counting(target: Path, output: Path, link_extension: str) -> str: - calls.append(target) - return relative_link(target, output, link_extension) + workspace = Path(str(error.value).split("cleanup was refused at ", 1)[1].split(";", 1)[0]) + assert not output.exists() + assert (workspace / "site/publication.json").read_bytes() == previous_manifest + assert (intended_stage / PUBLICATION_MANIFEST).read_bytes() != previous_manifest - monkeypatch.setattr(mermaid, "relative_link", counting) - return calls +def test_first_install_parent_swap_never_publishes_into_the_replacement_parent( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + publication_parent = tmp_path / "publication-parent" + publication_parent.mkdir() + destination = publication_parent / "out" + replacement_parent = tmp_path / "replacement-parent" + replacement_parent.mkdir() + sentinel = replacement_parent / "sentinel" + sentinel.write_text("keep\n", encoding="utf-8") + moved_parent = tmp_path / "original-parent" + original = render_module._rename_noreplace -def test_node_links_resolve_each_target_page_once(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: - """Building a link per node resolved paths on disk once per node on every - page, which made render time grow with the square of the node count.""" - from autoform_cli.render import _anchored_links + def swap_parent_then_install(*args): + publication_parent.rename(moved_parent) + replacement_parent.rename(publication_parent) + original(*args) - chapter, other, page = tmp_path / "a.md", tmp_path / "b" / "README.md", tmp_path / "page.md" - targets = {f"a/{index}": (chapter, f"n{index}") for index in range(20)} - targets["b"] = (other, "") - targets["here"] = (page, "self") - calls = _count_relative_links(monkeypatch) - resolved: list[Path] = [] - resolve = Path.resolve + monkeypatch.setattr(render_module, "_rename_noreplace", swap_parent_then_install) - def counting_resolve(self: Path, strict: bool = False) -> Path: - resolved.append(self) - return resolve(self, strict) + with pytest.raises(PublicationError, match="original output-parent generation"): + render_site(project / "blueprint", destination, lean_root=project) - monkeypatch.setattr(Path, "resolve", counting_resolve) + assert (publication_parent / "sentinel").read_text(encoding="utf-8") == "keep\n" + assert not destination.exists() + assert (moved_parent / "out/publication.json").is_file() + assert len(list(moved_parent.glob(f"{render_module._PUBLICATION_STAGE_PREFIX}*"))) == 1 - links = _anchored_links(targets, page) - assert sorted(calls) == [chapter, other] - # Once to compare it with the current page, once inside relative_link. - assert resolved.count(chapter) <= 2 - assert links["a/7"] == "a.html#n7" - assert links["b"] == "b/index.html" - assert links["here"] == "#self" +def test_replacement_parent_swap_never_overwrites_the_replacement_parent( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + publication_parent = tmp_path / "publication-parent" + publication_parent.mkdir() + destination = publication_parent / "out" + render_site(project / "blueprint", destination, lean_root=project) + previous_manifest = (destination / PUBLICATION_MANIFEST).read_bytes() + article = project / "blueprint/roadmap/top.md" + article.write_text( + article.read_text(encoding="utf-8") + "\nNew generation.\n", + encoding="utf-8", + ) + replacement_parent = tmp_path / "replacement-parent" + replacement_destination = replacement_parent / "out" + replacement_destination.mkdir(parents=True) + sentinel = replacement_destination / "sentinel" + sentinel.write_text("keep\n", encoding="utf-8") + moved_parent = tmp_path / "original-parent" + original = render_module._rename_exchange + def swap_parent_then_exchange(*args): + publication_parent.rename(moved_parent) + replacement_parent.rename(publication_parent) + original(*args) -def test_rewriting_a_page_links_only_the_nodes_it_names(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: - """A page uses the few node links it contains, so rewriting it should not - build a link for every node in the graph.""" - from autoform_cli.render import _rewrite_links + monkeypatch.setattr(render_module, "_rename_exchange", swap_parent_then_exchange) - blueprint, destination = tmp_path / "blueprint", tmp_path / "out" - source_dir = blueprint / "roadmap" - source_dir.mkdir(parents=True) - chapter, other, page = destination / "a.md", destination / "b.md", destination / "page.md" - targets = {"a/x": (chapter, "x"), "a/z": (chapter, "z"), "b/y": (other, "y")} - node_sources = { - (source_dir / "x.md").resolve(): "a/x", - (source_dir / "z.md").resolve(): "a/z", - (source_dir / "y.md").resolve(): "b/y", - } - calls = _count_relative_links(monkeypatch) + with pytest.raises(PublicationError, match="original output-parent generation"): + render_site(project / "blueprint", destination, lean_root=project) - def rewrite(text: str) -> str: - return _rewrite_links( - text, - source_dir=source_dir, - page=page, - blueprint=blueprint, - destination=destination, - node_sources=node_sources, - targets=targets, - ) + assert (destination / "sentinel").read_text(encoding="utf-8") == "keep\n" + assert (moved_parent / "out/publication.json").read_bytes() != previous_manifest + workspace = next(moved_parent.glob(f"{render_module._PUBLICATION_STAGE_PREFIX}*")) + assert (workspace / "site/publication.json").read_bytes() == previous_manifest - assert rewrite("Plain prose.\n") == "Plain prose.\n" - assert calls == [] - assert rewrite("See [X](x.md).\n") == "See [X](a.md#x).\n" - assert calls == [chapter] - calls.clear() - # Two nodes on the same chapter page cost one link to it, not one each. - assert rewrite("[X](x.md), [Z](z.md)\n") == "[X](a.md#x), [Z](a.md#z)\n" - assert calls == [chapter] +def test_explicit_symlink_lean_root_is_canonicalized_once(tmp_path: Path) -> None: + project = _project(tmp_path) + ref = _commit_project(project) + alias = tmp_path / "project-alias" + alias.symlink_to(project, target_is_directory=True) + output = tmp_path / "out" + + report = render_site( + project / "blueprint", + output, + lean_root=alias, + repository_url="https://github.com/owner/repo", + ref=ref, + ) + + assert report.linked == 2 + chapter = (output / "roadmap/README.md").read_text(encoding="utf-8") + assert f"/blob/{ref}/blueprint/roadmap/top.md" in chapter -def test_a_focus_page_asks_for_its_node_links_once(tmp_path: Path) -> None: - """Each request builds a link for every node, so asking twice per focus - page doubled the cost of the largest group of generated pages.""" - from autoform_cli.graph_pages import focus_page_path, write_graph_pages +def test_explicit_macos_var_alias_lean_root_is_canonicalized(tmp_path: Path) -> None: project = _project(tmp_path) - graph = load_graph(project / "blueprint") - destination = tmp_path / "out" - requested: list[Path] = [] + ref = _commit_project(project) + physical = str(project) + if not physical.startswith("/private/var/"): + pytest.skip("macOS /var alias is unavailable") + alias = Path("/var") / Path(physical).relative_to("/private/var") + if not alias.exists() or alias.resolve() != project.resolve(): + pytest.skip("macOS /var alias is unavailable") - def node_links(page: Path) -> dict[str, str]: - requested.append(page) - return {node_id: f"roadmap.html#{node_id}" for node_id in graph.nodes} + report = render_site( + project / "blueprint", + tmp_path / "out", + lean_root=alias, + repository_url="https://github.com/owner/repo", + ref=ref, + ) - write_graph_pages(graph, derive(graph), destination, node_links=node_links) + assert report.linked == 2 - for node_id in ("base", "top"): - page = focus_page_path(destination, node_id) - assert requested.count(page) == 1 - assert f"roadmap.html#{node_id}" in page.read_text(encoding="utf-8") +def test_cleanup_indexes_each_inventory_record_once( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + root = tmp_path / "cleanup-tree" + root.mkdir() + for index in range(200): + directory = root / f"directory-{index:03d}" + directory.mkdir() + (directory / "file.txt").write_text(str(index), encoding="utf-8") + inventory = render_module._cleanup_inventory(root) + expected_visits = len(inventory.directories) + len(inventory.files) + original = render_module.PurePosixPath + visits = 0 + + def count_path(value: str): + nonlocal visits + visits += 1 + return original(value) + + monkeypatch.setattr(render_module, "PurePosixPath", count_path) + descriptor = render_module._open_directory_path(root) + try: + render_module._remove_inventory_contents(descriptor, inventory) + finally: + os.close(descriptor) + + assert visits == expected_visits + assert not any(root.iterdir()) + + +def test_a_render_rereads_each_published_file_a_bounded_number_of_times( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + output = tmp_path / "out" + original = render_module._read_regular_file_at + reads = 0 -def test_a_node_that_is_the_current_page_links_as_a_bare_fragment(tmp_path: Path) -> None: - """A container article has no anchor of its own, and an empty href would - drop the link, so its own page links to the top of itself.""" - from autoform_cli.render import _anchored_links + def count(parent_descriptor, name, display_path, **kwargs): + nonlocal reads + if name == "top.md": + reads += 1 + return original(parent_descriptor, name, display_path, **kwargs) - page = tmp_path / "chapter" / "README.md" + monkeypatch.setattr(render_module, "_read_regular_file_at", count) + render_site(project / "blueprint", output, lean_root=project) + first_install, reads = reads, 0 + render_site(project / "blueprint", output, lean_root=project) - assert _anchored_links({"chapter": (page, ""), "chapter/x": (page, "x")}, page) == { - "chapter": "#", - "chapter/x": "#x", - } + # Each read is a full SHA-256 pass; these bounds catch new redundant passes. + assert first_install <= 7 + assert reads <= 16 -def _conditional_project(tmp_path: Path, policy: str) -> Path: - """`_project` with Top proved from an open statement, under the given policy.""" +def test_stage_directory_symlink_substitution_never_receives_plan_bytes( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: project = _project(tmp_path) - roadmap = project / "blueprint" / "roadmap" - (roadmap / "README.md").write_text( - f"---\nopen_statements: {policy}\n---\n\n# Roadmap\n\n" - "## Definitions\n\n- [Base](base.md)\n\n" - "## Results\n\n- [Open](open.md)\n- [Top](top.md)\n", - encoding="utf-8", - ) - (project / "Project" / "Open.lean").write_text( - "namespace Project\n\ntheorem openStatement : True := sorry\n\nend Project\n", encoding="utf-8" - ) - (roadmap / "open.md").write_text( - "---\ndeclaration: theorem\nstatement: formalized\nlean: Project.openStatement\n---\n\n" - "# Open\n\nA statement whose proof is still sorry.\n\n## Depends on\n\n- [Base](base.md)\n", - encoding="utf-8", - ) - top = roadmap / "top.md" - top.write_text( - top.read_text(encoding="utf-8") + "\n## Proof depends on\n\n- [Open](open.md)\n", encoding="utf-8" + output = tmp_path / "out" + foreign = tmp_path / "foreign-directory" + foreign.mkdir() + sentinel = foreign / "SENTINEL" + sentinel.write_text("untouched\n", encoding="utf-8") + substituted = False + + def substitute_directory(event: str, relative: str) -> None: + nonlocal substituted + if substituted or event != "after-directory-open" or relative != "roadmap": + return + workspace = next(tmp_path.glob(f"{render_module._PUBLICATION_STAGE_PREFIX}*")) + roadmap = workspace / "site/roadmap" + roadmap.rename(workspace / "site/roadmap-owned") + roadmap.symlink_to(foreign, target_is_directory=True) + substituted = True + + monkeypatch.setattr( + render_module, + "_publication_plan_checkpoint", + substitute_directory, ) - return project + with pytest.raises(PublicationError, match="workspace was retained"): + render_site(project / "blueprint", output, lean_root=project) -def test_a_conditional_proof_names_the_open_statements_it_assumes( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch -) -> None: - # The Lean row is the fallback for a declaration without a source link, and - # on CI the Actions environment would supply the coordinates for one. - for variable in ("GITHUB_REPOSITORY", "GITHUB_SERVER_URL", "GITHUB_SHA"): - monkeypatch.delenv(variable, raising=False) - project = _conditional_project(tmp_path, "allowed") + assert substituted + assert not output.exists() + assert sentinel.read_text(encoding="utf-8") == "untouched\n" + assert list(foreign.iterdir()) == [sentinel] - render_site(project / "blueprint", tmp_path / "out", lean_root=project, repository_url="", ref="") - page = (tmp_path / "out/roadmap/README.md").read_text(encoding="utf-8") - top = page[page.index('id="top"'):] - assert '
conditionally proved' in top - assert ( - 'Assumes' - 'Theorem 1 (Open)' - " (open statements without a recorded Lean proof)" - ) in top - # Only the conditional proof carries the row; the open statement assumes nothing. - assert page.count('Assumes') == 1 - # The row follows the implementation row and precedes Discussion. - meta = top[top.index('
'):top.index('
')] - assert re.findall(r'([^<]+)', meta) == ["Lean", "Assumes", "Discussion"] +def test_output_parent_loss_after_verified_commit_is_uncertain( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + publication_parent = tmp_path / "publication-parent" + publication_parent.mkdir() + output = publication_parent / "out" + replacement_parent = tmp_path / "replacement-parent" + replacement_parent.mkdir() + (replacement_parent / "sentinel").write_text("keep\n", encoding="utf-8") + moved_parent = tmp_path / "moved-output-parent" + original = render_module._publish_staged_site + + def move_parent_after_verified_commit(*args, **kwargs): + original(*args, **kwargs) + publication_parent.rename(moved_parent) + replacement_parent.rename(publication_parent) + + monkeypatch.setattr( + render_module, + "_publish_staged_site", + move_parent_after_verified_commit, + ) - css = (tmp_path / "out/stylesheets/blueprint.css").read_text(encoding="utf-8") - assert ".bp-ref-conditional::before, .bp-swatch-conditional { background: #E9DFFC; border-color: #6B3FCF; }" in css - assert "[data-md-color-scheme=slate] .bp-ref-conditional::before" in css - assert ".bp-conditional .bp-mark { color: #6B3FCF; }" in css + with pytest.raises(PublicationError, match="output location is uncertain") as error: + render_site(project / "blueprint", output, lean_root=project) + workspace = next(moved_parent.glob(f"{render_module._PUBLICATION_STAGE_PREFIX}*")) + assert workspace.name in str(error.value) + assert str(workspace) not in str(error.value) + assert (moved_parent / "out/publication.json").is_file() + assert (publication_parent / "sentinel").read_text(encoding="utf-8") == "keep\n" -def test_the_strict_policy_shows_the_same_proof_as_proved_with_no_assumptions(tmp_path: Path) -> None: - project = _conditional_project(tmp_path, "forbidden") - render_site(project / "blueprint", tmp_path / "out", lean_root=project) - page = (tmp_path / "out/roadmap/README.md").read_text(encoding="utf-8") +def test_output_parent_swap_during_cleanup_cannot_return_success( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + project = _project(tmp_path) + publication_parent = tmp_path / "publication-parent" + publication_parent.mkdir() + output = publication_parent / "out" + replacement_parent = tmp_path / "replacement-parent" + replacement_parent.mkdir() + (replacement_parent / "sentinel").write_text("keep\n", encoding="utf-8") + moved_parent = tmp_path / "moved-output-parent" + original = render_module._remove_owned_workspace + swapped = False + + def cleanup_then_swap(*args, **kwargs): + nonlocal swapped + cleaned = original(*args, **kwargs) + assert cleaned + publication_parent.rename(moved_parent) + replacement_parent.rename(publication_parent) + swapped = True + return cleaned + + monkeypatch.setattr(render_module, "_remove_owned_workspace", cleanup_then_swap) + + with pytest.raises(PublicationError, match="output location is uncertain") as error: + render_site(project / "blueprint", output, lean_root=project) - assert '
Assumes' not in page + assert swapped + assert "workspace cleanup completed" in str(error.value) + assert not output.exists() + assert (moved_parent / "out/publication.json").is_file() + assert not list(moved_parent.glob(f"{render_module._PUBLICATION_STAGE_PREFIX}*")) + assert (publication_parent / "sentinel").read_text(encoding="utf-8") == "keep\n" -@pytest.mark.parametrize( - ("dropped", "why"), - [ - ("proof: formalized\n", "Its prerequisites are ready, so the proof can be written now."), - ( - "statement: formalized\nproof: formalized\n", - "Its prerequisites are ready, so the statement can be written down.", - ), - ], -) -def test_next_up_explains_readiness_without_naming_a_policy(tmp_path: Path, dropped: str, why: str) -> None: +def test_precommit_parent_loss_does_not_report_a_false_workspace_path( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: project = _project(tmp_path) - top = project / "blueprint" / "roadmap" / "top.md" - top.write_text(top.read_text(encoding="utf-8").replace(dropped, ""), encoding="utf-8") - - render_site(project / "blueprint", tmp_path / "out", lean_root=project) - landing = (tmp_path / "out/README.md").read_text(encoding="utf-8") + publication_parent = tmp_path / "publication-parent" + publication_parent.mkdir() + output = publication_parent / "out" + replacement_parent = tmp_path / "replacement-parent" + replacement_parent.mkdir() + (replacement_parent / "sentinel").write_text("keep\n", encoding="utf-8") + moved_parent = tmp_path / "moved-output-parent" + + def lose_parent_before_commit(_descriptor: int) -> None: + publication_parent.rename(moved_parent) + replacement_parent.rename(publication_parent) + raise OSError("injected stage sync failure") + + monkeypatch.setattr(render_module, "_sync_tree_descriptor", lose_parent_before_commit) + + with pytest.raises(PublicationError, match="original output-parent generation") as error: + render_site(project / "blueprint", output, lean_root=project) - assert '
' in landing - assert f'
{why}
' in landing + workspace = next(moved_parent.glob(f"{render_module._PUBLICATION_STAGE_PREFIX}*")) + assert workspace.name in str(error.value) + assert str(publication_parent / workspace.name) not in str(error.value) + assert (publication_parent / "sentinel").read_text(encoding="utf-8") == "keep\n" def _statement_box(page: str, node_id: str) -> str: diff --git a/tests/test_scaffold.py b/tests/test_scaffold.py index b23e7c81..94de1f49 100644 --- a/tests/test_scaffold.py +++ b/tests/test_scaffold.py @@ -471,7 +471,7 @@ def test_scaffold_merges_autoform_rules_into_lake_gitignore(tmp_path: Path) -> N merged = (tmp_path / ".gitignore").read_bytes() assert merged.startswith(lake_ignore) - for rule in (b".lake/", b"site/", b"site-src/", b"*.log"): + for rule in (b".lake/", b"site/", b"site-src/", b".autoform-publication-*/", b"*.log"): assert merged.splitlines().count(rule) == 1 assert ".gitignore" in result.written assert ".gitignore" not in result.skipped diff --git a/tests/test_skill_examples.py b/tests/test_skill_examples.py index 7b5b52b9..11c5def3 100644 --- a/tests/test_skill_examples.py +++ b/tests/test_skill_examples.py @@ -2,6 +2,8 @@ import json import re +import shutil +import subprocess from pathlib import Path try: @@ -334,7 +336,33 @@ def test_roadmap_example_is_structural_not_a_completion_fixture( def test_setup_asset_static_site_contract(repo_root: Path, tmp_path: Path) -> None: - example = repo_root / _EXAMPLE + example = tmp_path / "consumer-project" + shutil.copytree(repo_root / _EXAMPLE, example) + subprocess.run(["git", "init", "-q"], cwd=example, check=True) + subprocess.run(["git", "add", "--all"], cwd=example, check=True) + subprocess.run( + [ + "git", + "-c", + "user.name=Autoform Test", + "-c", + "user.email=autoform@example.invalid", + "commit", + "-q", + "--no-gpg-sign", + "-m", + "consumer fixture", + ], + cwd=example, + check=True, + ) + ref = subprocess.run( + ["git", "rev-parse", "HEAD"], + cwd=example, + check=True, + capture_output=True, + text=True, + ).stdout.strip() site = tmp_path / "site-src" report = render_site( @@ -342,15 +370,15 @@ def test_setup_asset_static_site_contract(repo_root: Path, tmp_path: Path) -> No site, lean_root=example, repository_url="https://github.com/owner/repo", - ref="0" * 40, + ref=ref, ) assert report.unresolved == [] manifest = json.loads((site / "publication.json").read_text(encoding="utf-8")) - assert manifest["schema"] == "autoform-publication/v1" + assert manifest["schema"] == "autoform-publication/v2" assert manifest["nodes"] == 10 assert manifest["dependencies"] == 9 - assert manifest["git_ref"] == "0" * 40 + assert manifest["git_ref"] == ref assert manifest["coverage"]["complete"] is False assert manifest["coverage"]["counts"] == { "DECOMPOSED": 1, @@ -799,6 +827,9 @@ def test_cli_reference_documents_only_commands_that_exist(repo_root: Path) -> No from autoform_cli.__main__ import main reference = (repo_root / "autoform_cli/README.md").read_text(encoding="utf-8") + normalized_reference = " ".join(reference.split()) + assert "match one locally available Git commit" in normalized_reference + assert "records no Git ref" in normalized_reference documented = _documented_invocations(reference) assert {("check",), ("audit",), ("render",), ("search",), ("claim", "acquire")} <= documented diff --git a/tests/test_windows_source_boundary.py b/tests/test_windows_source_boundary.py index 7fbd35d9..859c6190 100644 --- a/tests/test_windows_source_boundary.py +++ b/tests/test_windows_source_boundary.py @@ -6,6 +6,7 @@ import pytest from autoform_cli.lean import index_project, snapshot_project_sources +from autoform_cli.render import PublicationError, render_site @pytest.mark.skipif(os.name != "nt", reason="Windows capability boundary") @@ -21,3 +22,20 @@ def test_windows_source_inspection_fails_closed_without_descriptor_traversal( with pytest.raises(OSError, match="safe directory traversal is unavailable"): capture(tmp_path) + + +@pytest.mark.skipif(os.name != "nt", reason="Windows capability boundary") +def test_windows_render_fails_before_writing_or_changing_the_output(tmp_path: Path) -> None: + blueprint = tmp_path / "blueprint" + blueprint.mkdir() + output = tmp_path / "site-src" + output.mkdir() + sentinel = output / "PRECIOUS" + sentinel.write_text("keep\n", encoding="utf-8") + + with pytest.raises(PublicationError, match="unavailable on this platform"): + render_site(blueprint, output) + + assert sentinel.read_text(encoding="utf-8") == "keep\n" + assert list(output.iterdir()) == [sentinel] + assert not list(tmp_path.glob(".autoform-publication-*"))