diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index c84111f..4fc4994 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -22,7 +22,7 @@ jobs: tests: uses: QuietFlare/ci-workflows/.github/workflows/python-test.yml@v1 with: - test-command: 'python3 -m unittest discover -s tests' + test-command: 'make dev test' # 3.9 is what macOS ships, so it is the floor people actually hit. python-versions: '["3.9", "3.11", "3.13"]' @@ -48,7 +48,7 @@ jobs: python-version: "3.12" - name: Install the driver - run: pip install -r requirements.txt + run: pip install -r requirements.txt && make dev - name: Run the full suite against Postgres env: diff --git a/CHANGELOG.md b/CHANGELOG.md index 99e7439..a3c7c20 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,23 @@ follow [Semantic Versioning](https://semver.org/). - ADR 0008: a bundle verifies against something it does not control. - ADR 0009: an effective date is an instant, and the gate has a now. +### Changed + +- One shipped policy table, `v1`, with five dimensions: `contribution`, + `storage`, `scope` (`exclusive` or `shared`), `released` and `mode` + (`remove` or `trace`). Release is asked before existence, and a corrected + subject's separable part is regenerated rather than purged (R5). The + earlier tables and the names `exclusive`, `terminal` and `because` are + gone. +- Plan items are `clew_plan_version` 2: `scope`, `released`, `mode`, + `reason` for the rule's rationale and `evidence` for how the class was + found. The printed plan carries a column header. +- `--assertions` files list released artifacts under `released`. +- `clew evidence build`: `--out` is optional and defaults to + `-`. +- The plan header says "removal of", not "withdrawal of". +- `clew providers` names each provider's package on Python 3.9 too. + ### Fixed - `clew impact`: evidence chains come from one breadth-first pass over the @@ -24,7 +41,6 @@ follow [Semantic Versioning](https://semver.org/). withheld. Directory outputs are now found under `--results` by name. - `clew impact`: a subject that matches no task tag exits non-zero instead of reporting zero affected tasks. -<<<<<<< HEAD - `clew reclaim`: a task is `FAILED` only when the engine recorded a failure. Every extractor now maps its engine's status word (`Done`, `SUCCEEDED`, `skipped`, `CACHED`) to one of `COMPLETED`, `FAILED`, @@ -92,7 +108,6 @@ follow [Semantic Versioning](https://semver.org/). are skipped by the bundle store. - `eventlog.append`: `recorded_at` is the database server's clock, read in the appending transaction. Callers can no longer supply it. -======= - `clew extract-work`: refuses when any task's work directory is missing, naming them; `--allow-partial` writes the graph with the gap as a `coverage` note. A cleaned tree used to give an empty graph and exit 0. @@ -127,7 +142,6 @@ follow [Semantic Versioning](https://semver.org/). donor table without `--subject` now reads the samplesheet. - `clew extract-work` refuses a six-character hash prefix that matches two work directories instead of merging them. ->>>>>>> f360682 (Attribute subjects longest first, surface graph limits in impact) ## [0.4.0] - 2026-09-07 diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index acee94b..24442e4 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -13,15 +13,14 @@ and it improves by contact with practitioners. After that, in rough order: -- A domain adapter for a pipeline you use. See `clew/domains/`. A new - adapter is usually a few dozen lines on top of `nfcore.py`, plus a - regression test pinning real numbers from a real run. +- A domain adapter for a pipeline you use, or an extractor for an engine + Clew cannot read. Both are one subclass in your own package, found by + name. [docs/providers.md](docs/providers.md) walks through each with + examples. A regression test pinning real numbers from a real run is + what makes an adapter trustworthy. - A bug report with a graph. A wrong blast radius is the most serious class of bug here, above all one that reports something as unaffected when it is not. Attach the graph JSON if you can share it. -- An extractor for another engine or provenance format. Every extractor - emits the same graph JSON, so `clew/extract/horus.py` is a good - model. ## Ground rules for code @@ -39,10 +38,21 @@ After that, in rough order: decides. An auditor asking why something was flagged must get a policy version, hashes, and a re-run that agrees. -Run the suite before opening a pull request: +The engine and the six providers under `providers/` are separate +distributions. Install all seven editable first, or nothing registers: ```bash -python3 -m unittest discover -s tests +make dev +``` + +Pass `PYTHON=` if the first `python3` on your PATH is not the one `clew` +runs under. + +Run every suite before opening a pull request. The engine's tests are in +`tests/`; each provider's are in its own `tests/`, beside its fixtures: + +```bash +make test ``` ## Licensing of contributions diff --git a/Makefile b/Makefile new file mode 100644 index 0000000..d89ae99 --- /dev/null +++ b/Makefile @@ -0,0 +1,16 @@ +# The engine and its providers are separate distributions. Install them all +# editable to work on the checkout; the tests need every provider present. +# PYTHON must be the interpreter clew runs under: make PYTHON=... if it is +# not the first python3 on PATH. +PYTHON ?= python3 +PROVIDERS := $(wildcard providers/*) + +dev: + $(PYTHON) -m pip install -e . $(addprefix -e ,$(PROVIDERS)) + +# The engine's suite, then each provider's, from its own tests/. +test: + $(PYTHON) -m unittest discover -s tests + @for p in $(PROVIDERS); do echo "== $$p"; $(PYTHON) -m unittest discover -s $$p/tests || exit 1; done + +.PHONY: dev test diff --git a/README.md b/README.md index 118cb00..a606b0e 100644 --- a/README.md +++ b/README.md @@ -20,9 +20,13 @@ A clew is the ball of thread Ariadne gave Theseus. You follow it back out. ## Install ```bash -pip install clew-lineage +pip install "clew-lineage[all]" ``` +That is the engine plus every provider it ships. `[nextflow]`, +`[snakemake]`, `[cromwell]`, `[horus]`, `[dnanexus]` or `[latch]` installs +one, and the bare `clew-lineage` is the engine alone. + ```bash clew demo ``` @@ -32,8 +36,8 @@ nf-core/sarek run that ships with the package. ## What it answers -**Something upstream went bad.** A reference update, a broken container, a -withdrawn sample. +**Something upstream went bad.** A reference update, a broken container, an +input that turned out wrong. ```bash clew impact --graph graph.json --container gatk4 @@ -82,12 +86,17 @@ One command per engine turns a run into a graph. | Engine | Command | |---|---| -| Nextflow, including Seqera Platform | `clew extract-store --store .lineage --run --json-out graph.json` | -| Snakemake | `clew extract-snakemake --workdir . --json-out graph.json` | -| Cromwell and WDL, including Terra | `clew extract-cromwell --metadata metadata.json --json-out graph.json` | -| Horus, through [horus-lineage](https://github.com/QuietFlare/horus-lineage) | `clew extract-horus --run-dir --json-out graph.json` | -| DNAnexus | `clew extract-dnanexus --analysis --json-out graph.json` | -| Latch | `clew extract-latch --execution --json-out graph.json` | +| Nextflow, including Seqera Platform | `clew extract nextflow --store .lineage --run --json-out graph.json` | +| Snakemake | `clew extract snakemake --workdir . --json-out graph.json` | +| Cromwell and WDL, including Terra | `clew extract cromwell --metadata metadata.json --json-out graph.json` | +| Horus, through [horus-lineage](https://github.com/QuietFlare/horus-lineage) | `clew extract horus --run-dir --json-out graph.json` | +| DNAnexus | `clew extract dnanexus --analysis --json-out graph.json` | +| Latch | `clew extract latch --execution --json-out graph.json` | + +`clew extract` lists every engine it knows, and `clew providers` shows +every adapter and extractor installed with the package each came from. +An engine or pipeline that is not there is one subclass away: see +[providers](docs/providers.md). Or skip the file: `reclaim`, `drift` and `digest` take `--runs` pointing at the engine's own record, a `.lineage` store or a horus-lineage root, @@ -108,7 +117,8 @@ no credentials. A gate can block a run before it starts. [Storage](docs/storage.md), [event log](docs/event-log.md), [policy](docs/policy.md), [evidence](docs/evidence.md), [gate](docs/gate.md), -[auditor surfaces](docs/auditors.md), [architecture](docs/architecture.md). +[auditor surfaces](docs/auditors.md), [architecture](docs/architecture.md), +[providers](docs/providers.md) for adding your own pipeline or engine. ## Status diff --git a/clew/__init__.py b/clew/__init__.py deleted file mode 100644 index 92005c3..0000000 --- a/clew/__init__.py +++ /dev/null @@ -1,10 +0,0 @@ -"""Clew rebuilds what pipeline runs derived from what, and answers questions over it.""" - -from clew.graph.blast_radius import blast_radius, load_graph -from clew.graph.contribution import classify -from clew.graph.triggers import parse as parse_trigger, resolve as resolve_trigger - -__version__ = "0.4.0" - -__all__ = ["blast_radius", "classify", "load_graph", "parse_trigger", - "resolve_trigger"] diff --git a/clew/__main__.py b/clew/__main__.py index 709e5f4..1675f52 100644 --- a/clew/__main__.py +++ b/clew/__main__.py @@ -13,7 +13,7 @@ "demo": ("clew.demo", "the shipped sample run: three triggers, one engine"), "impact": ("clew.questions.impact", - "what a withdrawal, defect or update reaches, and what to do"), + "what a removal, defect or update reaches, and what to do"), "gate": ("clew.questions.gate", "block a run whose inputs the log says are not usable"), "reclaim": ("clew.questions.reclaim", @@ -30,26 +30,26 @@ "one self-contained HTML page over sealed bundles"), "mcp": ("clew.views.mcp_server", "read-only MCP server over sealed bundles, for auditors"), + "extract": ("clew.extract", + "build a graph from an engine's record: clew extract "), + "providers": ("clew.providers", + "every domain and extractor installed, and the package each came from"), "stitch": ("clew.extract.stitch", "join run graphs where one run consumed another's outputs"), "digest": ("clew.extract.digest", "hash a run's files once, for graphs without content digests"), - "extract-store": ("clew.extract.nextflow_store", - "build a graph from the engine's native lineage store"), - "extract-crate": ("clew.extract.rocrate", - "build a graph from a Workflow Run RO-Crate"), - "extract-work": ("clew.extract.nextflow_work", - "build a graph from work/ symlinks, any engine version"), - "extract-horus": ("clew.extract.horus", - "build a graph from a horus-lineage run directory"), - "extract-dnanexus": ("clew.extract.dnanexus", - "build a graph from a DNAnexus analysis"), - "extract-latch": ("clew.extract.latch", - "build a graph from a Latch execution"), - "extract-cromwell": ("clew.extract.cromwell", - "build a graph from Cromwell workflow metadata"), - "extract-snakemake": ("clew.extract.snakemake", - "build a graph from Snakemake's metadata store"), +} + +# The names extractors had before `clew extract `. Still accepted. +ALIASES = { + "extract-store": "nextflow", + "extract-work": "nextflow-work", + "extract-crate": "ro-crate", + "extract-horus": "horus", + "extract-dnanexus": "dnanexus", + "extract-latch": "latch", + "extract-cromwell": "cromwell", + "extract-snakemake": "snakemake", } @@ -66,12 +66,14 @@ def usage(): def main(argv=None): argv = list(sys.argv[1:] if argv is None else argv) if argv and argv[0] in ("-V", "--version"): - from clew import __version__ - print(f"clew {__version__}") + from importlib.metadata import version + print(f"clew {version('clew-lineage')}") return 0 if not argv or argv[0] in ("-h", "--help"): print(usage()) return 0 + if argv[0] in ALIASES: + argv = ["extract", ALIASES[argv[0]]] + argv[1:] if argv[0] not in COMMANDS: print(f"clew: unknown command {argv[0]!r}\n\n{usage()}", file=sys.stderr) diff --git a/clew/contracts/__init__.py b/clew/contracts/__init__.py new file mode 100644 index 0000000..7e3e773 --- /dev/null +++ b/clew/contracts/__init__.py @@ -0,0 +1,8 @@ +"""from clew.contracts import Adapter, Extractor, Trigger""" + +from .adapter import Adapter +from .extractor import Extractor +from .registry import discover +from .trigger import REMOVE, TRACE, Mode, Trigger + +__all__ = ["Adapter", "Extractor", "Trigger", "Mode", "TRACE", "REMOVE", "discover"] diff --git a/clew/contracts/adapter.py b/clew/contracts/adapter.py new file mode 100644 index 0000000..9784736 --- /dev/null +++ b/clew/contracts/adapter.py @@ -0,0 +1,39 @@ +"""A domain is what a site knows about one pipeline. docs/providers.md has worked examples.""" + +from .registry import Provider + + +class Adapter(Provider): + group = "clew.adapters" + + # kind name -> Trigger. What can go wrong here, in this pipeline's own + # words, and where each enters. Empty is valid: the engine's own kinds + # still apply. + triggers = {} + + # reference files: triggers in their own right, never owned by anyone + load_bearing_inputs = () + + def contribution(self, graph, task_hash, kind): + """ + Optional. The class of this task's output with respect to one value of + `kind` being removed: SEPARABLE, REGENERABLE or IRREDUCIBLE. None keeps + the engine's evidence-based answer. The plan records that the adapter said so. + """ + return None + + def pending(self): + """ + Optional. Triggers recorded at this site and not yet asked: + [{"kind": "batch", "value": "B017", "asserted_by": "qa", "date": "2026-09-10"}] + With any returned, `clew impact --pipeline X` and no trigger answers each. + """ + return [] + + +def check(trigger): + """Problems with one trigger record; empty when usable.""" + if not isinstance(trigger, dict): + return ["not an object"] + return [f"{f} missing or empty" for f in ("kind", "value") + if not isinstance(trigger.get(f), str) or not trigger[f]] diff --git a/clew/contracts/extractor.py b/clew/contracts/extractor.py new file mode 100644 index 0000000..9dd73b9 --- /dev/null +++ b/clew/contracts/extractor.py @@ -0,0 +1,84 @@ +"""An extractor turns one engine's record of a run into the graph. The base owns what every engine shares.""" + +import argparse +import json +from abc import abstractmethod +from pathlib import Path + +from clew.graph.graph import EXTERNAL, contract_violations + +from .registry import Provider + + +class Extractor(Provider): + group = "clew.extractors" + description = "" + + @abstractmethod + def add_arguments(self, parser): + """This engine's source flags.""" + + @abstractmethod + def extract(self, args): + """The graph, or None when the command already answered (a listing).""" + + def records(self, path): + """ + Optional. If `path` is this engine's record, the runs in it: + {"root": Path, "runs": [{"name", "id", "timestamp", "session"?, "mtime"?}]} + `root` is where Clew keeps its sidecar. None when the path is not this engine's. + """ + return None + + def load(self, root, run_id): + """Optional, with records(): the graph of one run.""" + raise NotImplementedError(f"{self.name} does not load runs from a record") + + def summarize(self, graph, args): + known = set(graph["tasks"]) + edges = graph["edges"] + external = [e for e in edges if e["producer"] == EXTERNAL] + dangling = [e for e in edges + if e["producer"] not in known and e["producer"] != EXTERNAL] + print(f"tasks : {len(known)}") + print(f"input files (edges): {len(edges)}") + print(f" external inputs : {len(external)}") + print(f" DANGLING : {len(dangling)}") + if dangling: + print("\n=== DANGLING (producer not a task in this run) ===") + for e in dangling[:10]: + print(f"{e['consumer']} <- {e['producer']} ({e['filename']})") + self.coverage(graph) + + @staticmethod + def coverage(graph): + if graph.get("coverage"): + print("\n=== what this graph does not cover ===") + for note in graph["coverage"]: + print(f" - {note}") + + def parser(self): + parser = argparse.ArgumentParser(prog=f"clew extract {self.name}", + description=self.description) + self.add_arguments(parser) + parser.add_argument("--json-out", help="path to write the graph as JSON") + return parser + + def run(self, argv=None): + args = self.parser().parse_args(argv) + graph = self.extract(args) + if graph is None: + return 0 + problems = contract_violations(graph) + if problems: + raise SystemExit(f"clew extract {self.name}: the graph breaks the contract:\n " + + "\n ".join(problems[:20])) + self.summarize(graph, args) + if args.json_out: + Path(args.json_out).write_text(json.dumps(graph, indent=2)) + print(f"\nwrote {args.json_out}") + return 0 + + @classmethod + def main(cls, argv=None): + return cls.registered[cls.name].run(argv) diff --git a/clew/contracts/registry.py b/clew/contracts/registry.py new file mode 100644 index 0000000..1ea1d53 --- /dev/null +++ b/clew/contracts/registry.py @@ -0,0 +1,58 @@ +"""Defining a named subclass of a contract registers it. An entry point imports the module; that is all.""" + +from abc import ABC +from importlib import metadata + + +class Provider(ABC): + name = None + group = None + + def __init_subclass__(cls, **kwargs): + super().__init_subclass__(**kwargs) + if "registered" not in cls.__dict__ and Provider in cls.__bases__: + cls.registered = {} + if not cls.name: + return + root = next(c for c in cls.__mro__ if Provider in c.__bases__) + # ABCMeta has not marked this class yet, so check the root's set by hand. + missing = sorted(m for m in root.__abstractmethods__ + if getattr(getattr(cls, m), "__isabstractmethod__", False)) + if missing: + raise TypeError(f"{cls.__name__} does not implement {', '.join(missing)}") + root.registered[cls.name] = cls() + + +def entry_points(group): + """[(name, distribution name, entry point)] declared for a group, built-ins included.""" + # Walk distributions rather than metadata.entry_points(): an entry point + # only knows its distribution from 3.10, but a distribution has always + # known its entry points. One path for every supported Python. + listed, seen = [], set() + try: + dists = list(metadata.distributions()) + except Exception: # a broken distribution must not take the CLI down + return [] + for dist in dists: + try: + package, entries = dist.metadata["Name"], dist.entry_points + except Exception: + continue + for entry in entries: + if entry.group == group and (entry.name, entry.value) not in seen: + seen.add((entry.name, entry.value)) + listed.append((entry.name, package, entry)) + return listed + + +def discover(contract): + """{name: provider} for one contract, from every installed package.""" + declared = entry_points(contract.group) + for _, _, entry in declared: + entry.load() + if not declared and not contract.registered: + raise SystemExit( + f"no {contract.group} providers are installed. Clew's own are declared in " + "its package metadata, so a checkout must be installed: " + "python3 -m pip install -e .") + return dict(contract.registered) diff --git a/clew/contracts/trigger.py b/clew/contracts/trigger.py new file mode 100644 index 0000000..3eeefba --- /dev/null +++ b/clew/contracts/trigger.py @@ -0,0 +1,75 @@ +"""A trigger kind: how kind:value becomes entry nodes, and whether the value is traced or removed.""" + +from enum import Enum + +from clew.graph import triggers as engine + + +class Mode(Enum): + TRACE = "trace" # follow what it touched; everything stays, worst case quarantine + REMOVE = "remove" # the source is removed; what only it fed can be destroyed + + +TRACE, REMOVE = Mode.TRACE, Mode.REMOVE + + +class Trigger: + mode = Mode.TRACE + + def add_arguments(self, parser): + """Flags this kind needs, if any.""" + + def resolve(self, graph, value, args): + """{id: [entry nodes]}; a removal includes every peer, and value None means all.""" + raise NotImplementedError + + def values(self, args, graph=None): + """Every id this kind can name, for the gate. Optional.""" + raise NotImplementedError("this kind cannot list its values") + + +class FieldKind(Trigger): + """An engine kind: resolved from a field every graph carries.""" + + def __init__(self, name, finder): + self.name, self.finder = name, finder + + def resolve(self, graph, value, args): + if value is None: + raise SystemExit(f"{self.name}: a value is required, for example {self.name}:x") + return {f"{self.name}:{value}": self.finder(graph, value)} + + +class LabelKind(Trigger): + """Any other word: a label key the graph carries on tasks or edges.""" + + def __init__(self, key): + self.key = key + + def resolve(self, graph, value, args): + if value is None: + raise SystemExit(f"{self.key}: a value is required") + return {f"{self.key}:{value}": engine.label(self.key)(graph, value)} + + +ENGINE_KINDS = {name: FieldKind(name, finder) for name, finder in engine.KINDS.items()} + + +def lookup(domain, kind, graph=None): + """The domain's kind, else the engine's, else a label the graph carries, else None.""" + if domain is not None and kind in domain.triggers: + return domain.triggers[kind] + if kind in ENGINE_KINDS: + return ENGINE_KINDS[kind] + if graph is not None and kind in engine.label_keys(graph): + return LabelKind(kind) + return None + + +def parse(spec): + """`kind:value` or bare `kind`, which means every value of that kind.""" + kind, sep, value = spec.partition(":") + if not kind: + raise SystemExit(f"trigger {spec!r} should look like kind:value, " + "for example container:toolkit or batch:B017") + return kind, (value if sep else None) or None diff --git a/clew/data/assertions.json b/clew/data/assertions.json index b409a26..d63232e 100644 --- a/clew/data/assertions.json +++ b/clew/data/assertions.json @@ -1,6 +1,6 @@ { "_comment": "Externally-asserted facts the pipeline cannot know. Each record names WHO asserted WHAT and WHEN — these are inputs to Clew, never Clew's own claims. Synthetic example: the cohort QC report from this run was cited in a publication.", - "published": [ + "released": [ { "task": "c9/023b13", "what": "cohort MultiQC report, cited as Supplementary Fig 1", diff --git a/clew/demo.py b/clew/demo.py index a61c5e5..8fffca4 100644 --- a/clew/demo.py +++ b/clew/demo.py @@ -1,23 +1,16 @@ """ -Clew: the whole argument in one command, on one real pipeline run. +The whole argument in one command, on one real run. clew demo clew demo --work-root /path/to/work -Three questions, three audiences, one engine. Every number below is computed -live from graph5.json, a real nf-core/sarek run (5 synthetic donors, -81 tasks, 344 file-level edges) whose lineage was rebuilt from Nextflow's -work/ directory with no pipeline modification. - -WHAT THE DEMO CAN AND CANNOT SETTLE ------------------------------------ -The run's work/ was cleaned before it shipped, so the demo cannot check -whether any artifact is still on disk, and Clew never guesses. A verdict -that depends on storage is shown as OPEN, with the verdict each storage -state would produce, so the reader sees the whole answer short of the one -fact only a disk can supply. Pass --work-root on a run whose work/ still -exists and those lines settle. The published report settles without it: -under policy v2 publication is asked before existence. +Three questions, one engine, every number computed live from graph5.json: a +real nf-core/sarek run, 5 synthetic donors, 81 tasks, 344 edges, rebuilt +from work/ symlinks. The run's work/ was cleaned before it shipped, so a +verdict that depends on storage is shown OPEN with what each storage state +would settle to; --work-root on a live run settles them. The published +report settles without a disk: release is asked before +existence. """ import argparse @@ -26,12 +19,21 @@ from clew.graph import blast_radius as core +from clew.graph import graph as core_graph from clew.graph import contribution from clew.ledger import policy -from clew.domains import sarek +from clew.contracts import Adapter, discover ROOT = Path(__file__).resolve().parent + +def adapter(name): + """The shipped run is nf-core, so the demo needs the Nextflow provider.""" + try: + return discover(Adapter)[name] + except (KeyError, SystemExit): + raise SystemExit(f"clew demo needs the {name!r} adapter: pip install clew-nextflow") + OPEN = "OPEN" # How each storage state reads in a sentence. @@ -48,16 +50,14 @@ def plan_for(graph, affected, exclusive_set, published, work_root): """ - Verdict per affected task, grouped for display. - - Returns {(label, outcomes, why_open): [(hash, facts)]}. `label` is the - action, or OPEN when the verdict depends on storage. For OPEN groups - `outcomes` lists what each storage state would settle to, and - `why_open` says which fact is missing. + Verdict per affected task, grouped for display: {(label, outcomes, + why_open): [(hash, facts)]}. label is the action, or OPEN when it + depends on storage; outcomes then lists what each storage state would + settle to and why_open names the missing fact. """ plan = {} for task_hash in sorted(affected): - facts = sarek.classify(graph, task_hash, task_hash in exclusive_set, + facts = contribution.classify(graph, task_hash, task_hash in exclusive_set, published=published, work_root=work_root) why_open = NOT_CHECKED if facts["storage"] == contribution.DESTROYED: @@ -66,7 +66,7 @@ def plan_for(graph, affected, exclusive_set, published, work_root): # Leaving it open is the same rule clew impact applies. facts["storage"] = None why_open = CLEANED - dims = dict(exclusive=facts["exclusive"], terminal=facts["terminal"]) + dims = dict(scope=facts["scope"], released=facts["released"]) decision = policy.decide(facts["contribution"], storage=facts["storage"], **dims) if decision["action"]: @@ -93,9 +93,9 @@ def show(plan, graph, sample_rows=3): else: print(f" {label:<12} {len(rows):>3} {contribution.explain(label)}") for task_hash, facts in rows[:sample_rows]: - print(f" {task_hash} {sarek.describe(graph, task_hash)}") - if facts["terminal"]: - print(f" {facts['reason']}") + print(f" {task_hash} {core_graph.describe(graph, task_hash)}") + if facts["released"]: + print(f" {facts['evidence']}") if len(rows) > sample_rows: print(f" ... {len(rows) - sample_rows} more") @@ -111,8 +111,9 @@ def main(argv=None): work_root = args.work_root graph = core.load_graph(ROOT / "data" / "graph5.json") - donors = sarek.load_donors(ROOT / "data" / "donors.csv") - published = sarek.load_assertions(ROOT / "data" / "assertions.json") + patient = adapter("sarek").triggers["patient"] + donors = patient.ids(ROOT / "data" / "donors.csv") + published = core_graph.load_assertions(ROOT / "data" / "assertions.json") n = len(graph["tasks"]) print(f"Run: nf-core/sarek, {len(donors)} donors, {n} tasks, " @@ -129,7 +130,7 @@ def main(argv=None): print("=" * 70) print("1. ENGINEER: 'We bumped the reference genome. What must be re-run?'") print("=" * 70) - subjects = sarek.external_input_entry_nodes(graph, "genome.fasta") + subjects = core_graph.external_input_entry_nodes(graph, "genome.fasta") radius = core.blast_radius(graph, subjects) affected = radius["input:genome.fasta"]["affected"] entry = subjects["input:genome.fasta"] @@ -144,14 +145,14 @@ def main(argv=None): print("=" * 70) print("2. QA: 'A defect was reported in a GATK4 container. What did it touch?'") print("=" * 70) - subjects = sarek.container_entry_nodes(graph, "gatk4") + subjects = core_graph.container_entry_nodes(graph, "gatk4") radius = core.blast_radius(graph, subjects) affected = radius["container:gatk4"]["affected"] entry = subjects["container:gatk4"] print(f"\n {len(entry)} tasks ran in a gatk4 container; with everything") print(f" derived from their outputs: {len(affected)} of {n} tasks suspect.\n") show(plan_for(graph, affected, set(), published, work_root), graph) - print("\n Note: nothing can be DESTROYED here. A defect casts doubt; it") + print("\n Note: nothing can be DESTROYED here. A defect is traced; it") print(" does not remove a source. The artifacts are still wanted:") print(" rebuilt, not deleted. The published report is the one settled") print(" verdict, and it settles without a disk: publication outlives bytes.") @@ -161,7 +162,7 @@ def main(argv=None): print("=" * 70) print("3. COMPLIANCE: 'donor_003 withdrew consent. What happens now?'") print("=" * 70) - entry_by_donor = sarek.subject_entry_nodes(graph, donors) + entry_by_donor = patient.entries(graph, donors) radius = core.blast_radius(graph, entry_by_donor) r = radius["donor_003"] print(f"\n donor_003's material enters at {len(entry_by_donor['donor_003'])} tasks;" @@ -184,14 +185,13 @@ def main(argv=None): chain = ROOT / "data" / "graph_chain.json" sheet = ROOT / "data" / "samplesheets" / "rnaseq_yeast.csv" if chain.exists() and sheet.exists(): - from clew.domains import rnaseq - print() print("=" * 70) - print("4. THE CHAIN: one withdrawal, two pipelines") + print("4. THE CHAIN: one removal, two pipelines") print("=" * 70) g2 = core.load_graph(chain) - entry2 = rnaseq.subject_entry_nodes(g2, rnaseq.load_subjects(sheet)) + sample = adapter("rnaseq").triggers["sample"] + entry2 = sample.entries(g2, sample.ids(sheet)) radius2 = core.blast_radius(g2, entry2) r2 = radius2["SRR10441036_cox4d"] da = sorted(h for h in r2["affected"] if h.startswith("da:")) @@ -205,7 +205,7 @@ def main(argv=None): if g2["tasks"][h]["process"].endswith("DESEQ2_DIFFERENTIAL")) for path in core.paths_to(entry2["SRR10441036_cox4d"], target, forward2, limit=1): - hops = " -> ".join(f"{h}[{rnaseq.describe(g2, h)}]" for h in path) + hops = " -> ".join(f"{h}[{core_graph.describe(g2, h)}]" for h in path) print(f" evidence, crossing the run boundary:\n {hops}\n") print(" Engine-level lineage sees each launch in isolation. The") print(" crossing is the part only the stitched graph can answer.") @@ -218,7 +218,7 @@ def main(argv=None): print(f"Every verdict above is under policy {stamp['policy_version']}, " f"sha256 {stamp['policy_hash'][:16]};") print("`clew rulebook show` prints the table and the rationale for") - print("each rule; `clew rulebook diff v1 v2` shows what the last change to") + print("each rule; `clew rulebook diff v1 qbc.json` shows what a site changed against") print("it was, and why. `clew evidence build` seals any of the above into") print("a bundle that replays offline, and `clew gate` stops a run whose") print("inputs are not permitted before the pipeline starts.") diff --git a/clew/domains/nfcore.py b/clew/domains/nfcore.py deleted file mode 100644 index c2f06ce..0000000 --- a/clew/domains/nfcore.py +++ /dev/null @@ -1,197 +0,0 @@ -""" -Clew domain helpers shared by every nf-core pipeline adapter. - -nf-core pipelines share launch and naming conventions: a CSV samplesheet -with one row per sample, and task display names that append the sample tag -in parentheses ("BWAMEM1_MEM (donor_003)", "BOWTIE2_ALIGN (ERR10000000)"). -That convention, not anything pipeline-specific, is what Clew's subject -attribution parses — so it lives here, once. - -A pipeline adapter (sarek.py, viralrecon.py, rnaseq.py) supplies only what -actually differs: - - which samplesheet column names the subject (sarek: patient; most - others: sample) - - which external inputs are known to be load-bearing for that pipeline - -This file is still domain code: it may know about pipelines and samples. -core/ may not. -""" - -import csv - -from clew.graph import contribution -from clew.graph import graph as core_graph -import re -from pathlib import Path - -# The trailing parenthetical on a task's display name. Plenty of non-sample -# tasks use the same shape — "(genome)", "(MN908947.3)" — so a captured value -# is only accepted if it matches a subject from the samplesheet. -TAG_PATTERN = re.compile(r"\(([^()]+)\)\s*$") - -# Outputs every nf-core task writes for bookkeeping. They are collected -# through a channel rather than consumed as files, so no edge names them -# and no publishDir copies them one by one. -BOOKKEEPING = ("versions.yml",) - -# Index key standing in for a directory's size, which is meaningless. -DIRECTORY = "directory" - - -def load_subjects(samplesheet_path, subject_column, member_column=None): - """ - Read an nf-core samplesheet and return {subject_id: [member_ids]}. - - `subject_column` names the subject (sarek: "patient", most pipelines: - "sample"). `member_column`, when given, collects the per-subject members - (sarek: samples per patient); otherwise each subject stands alone. - """ - subjects = {} - with open(samplesheet_path, newline="") as handle: - for row in csv.DictReader(handle): - subject = (row.get(subject_column) or "").strip() - if not subject: - continue - subjects.setdefault(subject, []) - if member_column: - member = (row.get(member_column) or "").strip() - if member and member not in subjects[subject]: - subjects[subject].append(member) - # One id, one owner. The pipeline accepts the same member id under two - # subjects; attribution cannot, because whichever subject was read last - # would silently take the other's tasks. - owners = {} - shared = {} - for subject, members in subjects.items(): - for label in [subject] + members: - if label in owners and owners[label] != subject: - shared.setdefault(label, {owners[label]}).add(subject) - owners.setdefault(label, subject) - if shared: - listing = "; ".join(f"{label!r} under {', '.join(sorted(owners))}" - for label, owners in sorted(shared.items())) - raise SystemExit( - f"samplesheet {samplesheet_path}: the same id appears under " - f"more than one subject, so tasks tagged with it cannot be " - f"attributed: {listing}") - return subjects - - -def task_tag(task): - """Pull the trailing parenthetical off a task's display name, if present.""" - match = TAG_PATTERN.search(task.get("name", "")) - return match.group(1).strip() if match else None - - -def owner_of(tag, label_to_subject): - """ - Resolve a task tag to a subject, tolerating nf-core suffixes. - - Exact matching is not enough: per-lane steps append the lane - ("donor_003-L1", "ERR10000000_T1"). Prefixes are only accepted at a - separator boundary, so "donor_1" does not swallow "donor_10". Longest - label first, so "KO" does not claim "KO_2_T1" when "KO_2" is a label. - """ - if tag in label_to_subject: - return label_to_subject[tag] - for label in sorted(label_to_subject, key=len, reverse=True): - for separator in ("-", "_", "."): - if tag.startswith(label + separator): - return label_to_subject[label] - return None - - -def subject_entry_nodes(graph, subjects): - """ - Map each subject to the task nodes where its material enters the graph. - - Core sees only opaque ids. We start from EVERY task tagged with the - subject: over-inclusive is the safe direction — reporting a clean - artifact as affected wastes work, reporting an affected artifact as - clean is the failure that matters. - """ - label_to_subject = {} - for subject, members in subjects.items(): - label_to_subject[subject] = subject - for member in members: - label_to_subject[member] = subject - - entry = {subject: set() for subject in subjects} - for task_hash, task in graph["tasks"].items(): - tag = task_tag(task) - if not tag: - continue - owner = owner_of(tag, label_to_subject) - if owner: - entry[owner].add(task_hash) - - return {subject: sorted(nodes) for subject, nodes in entry.items()} - - -# --------------------------------------------------------------------------- -# Moved to core: none of these read anything nf-core specific, they only read -# the graph schema. Re-exported here so existing callers keep working. -# --------------------------------------------------------------------------- -container_entry_nodes = core_graph.container_entry_nodes -external_input_entry_nodes = core_graph.external_input_entry_nodes -load_assertions = core_graph.load_assertions -outputs_for = core_graph.outputs_for -describe = core_graph.describe -storage_state = contribution.storage_state -classify = contribution.classify - - -def index_results(results_dir): - """ - Index a published-results tree by (basename, size), plus directories - by (basename, DIRECTORY). - - Why this key: the artifacts in results/ are COPIES made by publishDir. - The lineage store's checksums are Nextflow's "standard" mode — hashed - from path and mtime — so they change on copy and cannot identify one. - Basename plus exact size can, almost always; where several published - files collide on both, every candidate is listed and the match is - flagged ambiguous rather than silently picking one. A deletion list - must over-report candidates, never guess. - - Directory outputs (a Salmon quant directory, a QualiMap report) have no - meaningful size, so they are indexed by name alone and any match is - flagged for verification. Missing them read a cleaned scratch as - ALREADY_GONE while the published directory was still on disk. - """ - index = {} - root = Path(results_dir) - for path in root.rglob("*"): - rel = str(path.relative_to(root)) - if path.is_file(): - index.setdefault((path.name, path.stat().st_size), []).append(rel) - elif path.is_dir(): - index.setdefault((path.name, DIRECTORY), []).append(rel) - return index - - -def published_copies(graph, task_hash, results_index): - """Published copies of one task's outputs, from the (name, size) index.""" - if not results_index: - return [] - matches = [] - for detail in graph.get("output_details", {}).get(task_hash, []): - name = Path(detail["file"]).name - found = results_index.get((name, detail.get("size")), []) - if found: - matches.append({ - "output": detail["file"], - "published": sorted(found), - "ambiguous": len(found) > 1, - }) - continue - found = results_index.get((name, DIRECTORY), []) - if found: - matches.append({ - "output": detail["file"], - "published": sorted(found), - # Name only; a directory's size says nothing. Over-report. - "ambiguous": True, - "match": "directory name", - }) - return matches diff --git a/clew/domains/rnaseq.py b/clew/domains/rnaseq.py deleted file mode 100644 index 3181667..0000000 --- a/clew/domains/rnaseq.py +++ /dev/null @@ -1,37 +0,0 @@ -""" -Clew domain adapter — nf-core/rnaseq (bulk RNA sequencing). - -The subject is the SAMPLE. The headline invalidation trigger here is the -ANNOTATION: gene models come from the GTF, so an annotation release bump -("Ensembl updated the GTF") invalidates every count matrix and differential -expression result computed against the old one — while the raw alignments -against the unchanged genome sequence may survive. That split is exactly -what per-input blast radii are for. -""" - -from . import nfcore - -# Basenames as recorded in a real 3.26.0 run's lineage store (iGenomes -# R64-1-1). Note the GTF: one direct consumer, yet 149 of 171 tasks in its -# blast radius — the annotation flows through a single preprocessing step -# and then touches nearly everything. Shallow entry, deep reach. -LOAD_BEARING_INPUTS = ( - "genome.fa", # iGenomes genome sequence - "genes.gtf", # annotation — the frequent-update trigger - "genes.bed", # gene models in BED form, consumed by the QC stack -) - - -def load_subjects(samplesheet_path): - """{sample: []} — each sample stands alone.""" - return nfcore.load_subjects(samplesheet_path, "sample") - - -task_tag = nfcore.task_tag -subject_entry_nodes = nfcore.subject_entry_nodes -container_entry_nodes = nfcore.container_entry_nodes -external_input_entry_nodes = nfcore.external_input_entry_nodes -load_assertions = nfcore.load_assertions -outputs_for = nfcore.outputs_for -describe = nfcore.describe -classify = nfcore.classify diff --git a/clew/domains/sarek.py b/clew/domains/sarek.py deleted file mode 100644 index 8de7953..0000000 --- a/clew/domains/sarek.py +++ /dev/null @@ -1,54 +0,0 @@ -""" -Clew domain adapter — nf-core/sarek. - -What is actually sarek-specific, after the shared nf-core conventions moved -to nfcore.py: the samplesheet models a DONOR ("patient") who can contribute -several samples (normal and tumour), so the subject is the patient column -and samples are its members. Donor identity is an ASSERTION carried in from -the samplesheet, never derived from file contents — five different donors' -alignment tasks all consumed files named test_1.fastq.gz on a real run. - -Known load-bearing external inputs for this pipeline: the reference bundle. -Invalidating any of them reaches everything calibrated against it. -""" - -from . import nfcore - -# Externals that must NOT propagate a donor withdrawal (they belong to no -# donor), but each is itself a valid invalidation trigger via --input. -LOAD_BEARING_INPUTS = ( - "genome.fasta", - "genome.fasta.fai", - "genome.dict", - "dbsnp_146.hg38.vcf.gz", - "mills_and_1000G.indels.vcf.gz", -) - - -def load_subjects(samplesheet_path): - """{donor: [sample_ids]} — `patient` is the donor; samples are members.""" - return nfcore.load_subjects(samplesheet_path, "patient", member_column="sample") - - -# Backwards-compatible name; blast.py and the tests grew up with it. -load_donors = load_subjects - -# Shared nf-core machinery, re-exported so callers need only this module. -task_tag = nfcore.task_tag -subject_entry_nodes = nfcore.subject_entry_nodes -container_entry_nodes = nfcore.container_entry_nodes -external_input_entry_nodes = nfcore.external_input_entry_nodes -load_assertions = nfcore.load_assertions -outputs_for = nfcore.outputs_for -describe = nfcore.describe -classify = nfcore.classify - - -def _owner_of(tag, label_to_donor): - """Kept under its old name for the regression tests.""" - return nfcore.owner_of(tag, label_to_donor) - - -def contribution_storage(task, work_root=None): - """Kept for compatibility; the logic lives in nfcore.storage_state.""" - return nfcore.storage_state(task, work_root) diff --git a/clew/domains/snakemake.py b/clew/domains/snakemake.py deleted file mode 100644 index c06c6b6..0000000 --- a/clew/domains/snakemake.py +++ /dev/null @@ -1,60 +0,0 @@ -""" -Clew domain adapter for Snakemake workflows. - -Snakemake has no sample tag. A job is named by its rule and first output -path, "trim (trimmed/sample_1.fq)", so the only place a sample id shows -is inside that path. This adapter reads the ids from a samplesheet column -and looks for each one in the output path: as a whole path component, or -as a token bounded by the separators file names are built from. That is -what keeps "sample_1" out of "sample_10.fq". - -Over-inclusion is the safe direction. A merge whose output is named after -two samples enters the graph for both. -""" - -import re - -from . import nfcore - -SEPARATORS = "-_./" - - -def load_subjects(samplesheet_path, column="sample"): - """{sample id: []} from one samplesheet column.""" - return nfcore.load_subjects(samplesheet_path, column) - - -def mentions(path, subject): - """Whether `subject` appears in `path` bounded by separators or the ends.""" - pattern = (rf"(?:^|[{re.escape(SEPARATORS)}])" - rf"{re.escape(subject)}" - rf"(?:$|[{re.escape(SEPARATORS)}])") - return re.search(pattern, path) is not None - - -def subject_entry_nodes(graph, subjects): - """ - {subject: [task hashes]} for every task whose tagged output path names - the subject. Ids are tried longest first; the separator bound in - `mentions` is what stops a shorter id matching inside a longer one. - """ - entry = {subject: set() for subject in subjects} - ordered = sorted(subjects, key=lambda s: (-len(s), s)) - for task_hash, task in graph["tasks"].items(): - tag = nfcore.task_tag(task) - if not tag: - continue - for subject in ordered: - if mentions(tag, subject): - entry[subject].add(task_hash) - return {subject: sorted(nodes) for subject, nodes in entry.items()} - - -# Shared machinery, re-exported so callers need only this module. -task_tag = nfcore.task_tag -container_entry_nodes = nfcore.container_entry_nodes -external_input_entry_nodes = nfcore.external_input_entry_nodes -load_assertions = nfcore.load_assertions -outputs_for = nfcore.outputs_for -describe = nfcore.describe -classify = nfcore.classify diff --git a/clew/domains/viralrecon.py b/clew/domains/viralrecon.py deleted file mode 100644 index 0881ba1..0000000 --- a/clew/domains/viralrecon.py +++ /dev/null @@ -1,43 +0,0 @@ -""" -Clew domain adapter — nf-core/viralrecon (viral surveillance). - -The subject is the SAMPLE (a sequenced specimen, e.g. an ENA run accession -like ERR10000000): the samplesheet has no donor concept, and none is -invented. The everyday invalidation trigger in this domain is not consent — -it is a specimen found contaminated or swapped after its data was used, and -reference or primer-scheme updates. - -Known load-bearing external inputs: the viral reference genome, its -annotation, and the amplicon primer scheme. A primer-scheme correction -invalidates every consensus built with it — the classic "quiet revision -with loud consequences". -""" - -from . import nfcore - -# Basenames as they appear in a real 2.6.0 run's lineage store (219-task -# COG-UK run). The same files carry different names in other releases — -# which is itself the argument for matching what the store recorded, not -# what a config file promises. -LOAD_BEARING_INPUTS = ( - "nCoV-2019.reference.fasta", # SARS-CoV-2 reference — 49 direct consumers - "nCoV-2019.primer.bed", # ARTIC primer scheme — the frequent-update trigger - "GCA_009858895.3_ASM985889v3_genomic.200409.gff.gz", # annotation - "kraken2_human.tar.gz", # host-removal database - "nextclade_sars-cov-2_MN908947_2022-06-14T12_00_00Z.tar.gz", # clade-call dataset -) - - -def load_subjects(samplesheet_path): - """{sample: []} — each specimen stands alone.""" - return nfcore.load_subjects(samplesheet_path, "sample") - - -task_tag = nfcore.task_tag -subject_entry_nodes = nfcore.subject_entry_nodes -container_entry_nodes = nfcore.container_entry_nodes -external_input_entry_nodes = nfcore.external_input_entry_nodes -load_assertions = nfcore.load_assertions -outputs_for = nfcore.outputs_for -describe = nfcore.describe -classify = nfcore.classify diff --git a/clew/extract/__init__.py b/clew/extract/__init__.py index e69de29..5ecb3f4 100644 --- a/clew/extract/__init__.py +++ b/clew/extract/__init__.py @@ -0,0 +1,35 @@ +""" +Extractors. Each is an Extractor from clew.contracts, declared in pyproject.toml like any provider. + + clew extract [flags] --json-out graph.json +""" + +import sys + +from clew.contracts import Extractor, discover + + +def registry(): + return discover(Extractor) + + +def usage(extractors): + width = max(len(name) for name in extractors) + lines = ["usage: clew extract [options]", "", "engines:"] + for name in sorted(extractors): + lines.append(f" {name.ljust(width)} {extractors[name].description}") + lines += ["", "clew extract --help shows that engine's options."] + return "\n".join(lines) + + +def main(argv=None): + argv = list(sys.argv[1:] if argv is None else argv) + extractors = registry() + if not argv or argv[0] in ("-h", "--help"): + print(usage(extractors)) + return 0 + if argv[0] not in extractors: + print(f"clew extract: unknown engine {argv[0]!r}\n\n{usage(extractors)}", + file=sys.stderr) + return 2 + return extractors[argv[0]].run(argv[1:]) diff --git a/clew/extract/runs.py b/clew/extract/runs.py index 6f21cd4..6fb6fb2 100644 --- a/clew/extract/runs.py +++ b/clew/extract/runs.py @@ -4,7 +4,7 @@ import sys from pathlib import Path -from clew.extract import horus, nextflow_store +from clew.contracts import Extractor, discover SIDECAR_DIR = ".clew" @@ -29,54 +29,47 @@ def recorded_timestamp(path): class Runs: + """ + An engine's record on disk. The extractor that recognises the path + lists its runs and loads one; a directory of graph JSON files needs + no extractor. Everything else here is engine-neutral: ordering, + resolving a name, the sidecar. + """ + def __init__(self, path): self.path = Path(path) - if (self.path / ".lineage").is_dir(): - self.path = self.path / ".lineage" - self.kind = self._detect() - - def _detect(self): - if (self.path / ".history").is_dir(): - return "nextflow" - if (self.path / horus.PLAN).is_file(): - return "horus-run" - if any((child / horus.PLAN).is_file() for child in self.path.iterdir() if child.is_dir()): - return "horus" - if any(child.suffix == ".json" for child in self.path.iterdir()): - return "graphs" - raise SystemExit(f"{self.path} is not a .lineage store, a horus-lineage root, " - "or a directory of graph JSON files") + self.extractor = None + for extractor in discover(Extractor).values(): + found = extractor.records(self.path) + if found is not None: + self.extractor, self.root = extractor, Path(found["root"]) + self.kind = extractor.name + self._runs = found["runs"] + return + if self.path.is_dir() and any(c.suffix == ".json" for c in self.path.iterdir()): + self.kind, self.root = "graphs", self.path + self._runs = [{"name": c.stem, "id": c.stem, + "timestamp": recorded_timestamp(c), + "mtime": c.stat().st_mtime} + for c in self.path.iterdir() if c.suffix == ".json"] + return + raise SystemExit(f"{self.path} is not an engine record any installed extractor " + "recognises, or a directory of graph JSON files") def records(self): - """ - [{name, id, timestamp, session, by_mtime}] oldest first. A run - whose record carries no timestamp is ordered by file mtime and - says so, since a copy or a touch reorders those. - """ - if self.kind == "nextflow": - return [{"name": r["name"], "id": r["run_hash"], "timestamp": r["timestamp"], - "session": r["session_id"], "by_mtime": False} - for r in nextflow_store.load_history(self.path)] - if self.kind == "horus-run": - return [{"name": self.path.name, "id": self.path.name, - "timestamp": recorded_timestamp(self.path / horus.PLAN), - "session": None, "by_mtime": False}] - if self.kind == "horus": - entries = [(c, c / horus.PLAN) for c in self.path.iterdir() - if (c / horus.PLAN).is_file()] - else: - entries = [(c, c) for c in self.path.iterdir() if c.suffix == ".json"] + """[{name, id, timestamp, session, by_mtime}] oldest first.""" found = [] - for entry, record in entries: - stamp = recorded_timestamp(record) - name = entry.stem if entry.is_file() else entry.name - found.append(({"name": name, "id": name, "timestamp": stamp, - "session": None, "by_mtime": not stamp}, - entry.stat().st_mtime)) + for r in self._runs: + stamp = r.get("timestamp") or "" + found.append({"name": r["name"], "id": r["id"], "timestamp": stamp, + "session": r.get("session"), "by_mtime": not stamp, + "_mtime": r.get("mtime", 0)}) # Timestamps and mtimes do not compare, so recorded ones sort # among themselves and the rest fall in by mtime after them. - found.sort(key=lambda pair: (pair[0]["by_mtime"], pair[0]["timestamp"], pair[1])) - return [record for record, _ in found] + found.sort(key=lambda r: (r["by_mtime"], r["timestamp"], r["_mtime"])) + for r in found: + del r["_mtime"] + return found def names(self): """[(name, id, timestamp)] oldest first.""" @@ -109,66 +102,50 @@ def resolve(self, wanted=None): + ", ".join(r["name"] for r in records)) return matches[0]["name"], matches[0]["id"] + def session_of(self, run_id): + return next((r["session"] for r in self.records() if r["id"] == run_id), None) + def load(self, wanted=None): """The graph of one run, with any sidecar digests merged in.""" name, run_id = self.resolve(wanted) - session = None - if self.kind == "nextflow": - run = nextflow_store.pick_run(nextflow_store.load_history(self.path), run_id) - session = run["session_id"] - graph = nextflow_store.extract(self.path, session) - elif self.kind == "horus-run": - graph = horus.extract(self.path) - elif self.kind == "horus": - graph = horus.extract(self.path / run_id) + if self.extractor: + graph = self.extractor.load(self.root, run_id) else: - graph = json.loads((self.path / f"{run_id}.json").read_text()) + graph = json.loads((self.root / f"{run_id}.json").read_text()) graph["run"] = {"name": name, "id": run_id} for sidecar in self.sidecar_paths(run_id): if sidecar.is_file(): merge_sidecar(graph, json.loads(sidecar.read_text())) + session = self.session_of(run_id) if session: # The graph is the chain's, not the run's: two runs of one # session load the same graph, and a caller comparing them # needs to know that. graph["run"]["session"] = session - sidecar = self.sidecar_path(run_id) - if sidecar.is_file(): - merge_sidecar(graph, json.loads(sidecar.read_text())) return graph - def sidecar_paths(self, run_id): + def sidecar_key(self, run_id): """ - Every sidecar that may hold this run's digests: the one filed under - the current key first, then any filed under a run hash of the same - chain by an earlier Clew, so digests already on disk keep counting. + A resumed chain's digests belong to the session, not to whichever + run name was typed; a digest written under one must be found under + the other. Engines without sessions key by run. """ + return self.session_of(run_id) or run_id + + def sidecar_paths(self, run_id): + """The sidecar under the current key, then any an earlier Clew filed under a run of the same chain.""" paths = [self.sidecar_path(run_id)] - if self.kind == "nextflow": - history = nextflow_store.load_history(self.path) - session = nextflow_store.pick_run(history, run_id)["session_id"] - for run in history: - if run["session_id"] == session: - legacy = self.path / SIDECAR_DIR / f"{run['run_hash']}.digests.json" + session = self.session_of(run_id) + if session: + for r in self.records(): + if r["session"] == session: + legacy = self.root / SIDECAR_DIR / f"{r['id']}.digests.json" if legacy not in paths: paths.append(legacy) return paths - def sidecar_key(self, run_id): - """ - What a sidecar is filed under. A store graph is the whole resume - chain, so its digests belong to the session, not to whichever run - name was typed; a digest written under one name must be found - under the other. - """ - if self.kind == "nextflow": - run = nextflow_store.pick_run(nextflow_store.load_history(self.path), run_id) - return run["session_id"] - return run_id - def sidecar_path(self, run_id): - base = self.path.parent if self.kind == "horus-run" else self.path - return base / SIDECAR_DIR / f"{self.sidecar_key(run_id)}.digests.json" + return self.root / SIDECAR_DIR / f"{self.sidecar_key(run_id)}.digests.json" def save_sidecar(self, graph): """Keep the sha256 digests of a graph beside the engine's record.""" diff --git a/clew/extract/stitch.py b/clew/extract/stitch.py index 2cbe2e7..496cbeb 100644 --- a/clew/extract/stitch.py +++ b/clew/extract/stitch.py @@ -27,13 +27,10 @@ def pre(h): def stitch(labelled_graphs): """ - Merge prefixed graphs and rewrite EXTERNAL edges whose digest another - run produced. Returns (graph, bridges). - - When several tasks in other runs produced the same digest, the edge is - bridged to every one of them: the consumer read those bytes, and which - task wrote them cannot be told apart by content. Choosing one would - drop the others from every blast radius. + Merge prefixed graphs and bridge EXTERNAL edges whose digest another run + produced. Returns (graph, bridges). Every producer of one digest gets + the edge: content cannot tell them apart, and picking one would drop the + rest from every blast radius. """ merged = {"tasks": {}, "edges": [], "outputs": {}, "output_details": {}} prefixed = {label: prefix_graph(label, g) for label, g in labelled_graphs.items()} diff --git a/clew/graph/blast_radius.py b/clew/graph/blast_radius.py index 07df7bf..75134e1 100644 --- a/clew/graph/blast_radius.py +++ b/clew/graph/blast_radius.py @@ -1,26 +1,10 @@ """ -Clew core — blast radius. +Blast radius: from a set of start nodes, everything downstream. -Given a graph and a set of starting nodes, find everything downstream. - -THIS FILE KNOWS NOTHING ABOUT BIOLOGY. -No wet-lab or workflow-engine vocabulary of any kind. It sees opaque node ids -and typed edges. Everything domain-specific lives in domains/. (CLAUDE.md -greps this directory for the forbidden words; even naming them here would -trip the check, which is why this sentence is vague on purpose.) - -If you ever need to write one of those words here, the design is wrong: the -knowledge belongs in a domain adapter that translates before calling in. - -DIRECTION ---------- -The extractor records edges BACKWARDS, because that is how a filesystem stores -them: a consumer holds a pointer to its producer. - - edge = {consumer: B, producer: A} meaning "B was made from A" - -A withdrawal travels FORWARDS: something at the source is revoked, and we need -everything built on top of it. So we invert the edges before traversing. +Opaque node ids and typed edges only; vocabulary lives in providers. The +extractor records edges backwards, consumer to producer, because that is how +a filesystem stores them. A removal travels forwards, so the edges are +inverted before traversal. """ import json @@ -74,17 +58,10 @@ def reachable(start_nodes, forward): def evidence_tree(start_nodes, forward): """ - One breadth-first pass from every start node: {node: parent}, with - parent None for the start nodes themselves. - - Computed once per trigger and read once per affected task. The earlier - per-target depth-first walk kept no visited set, so every scatter-gather - stage multiplied the paths it had to enumerate; fifty subjects with - three interval stages did not finish. This is linear in edges. - - Sorted, not incidental set order: which chain is recorded must not depend - on the interpreter's hash seed, because re-running on the same inputs - has to give byte-identical output. + One breadth-first pass from every start node: {node: parent}, None for + the start nodes. Linear in edges; the earlier per-target depth-first + walk did not finish on fifty subjects. Sorted, so the recorded chain + does not depend on the hash seed. """ parent = {node: None for node in start_nodes} queue = deque(sorted(start_nodes)) @@ -120,28 +97,11 @@ def paths_to(start_nodes, target, forward, limit=1): def blast_radius(graph, subjects): """ - Core entry point. - - `subjects` maps an opaque subject id to the nodes where that subject's - material enters the graph: - - {"subject-a": ["80/10c05c"], "subject-b": ["9b/e1fa4d"], ...} - - Core does not know or care what a subject is. The domain adapter decides. - - Returns, for each subject: - affected every node reachable from that subject - exclusive reachable from this subject and NO other - shared reachable from this subject and at least one other - - WHY THE SPLIT MATTERS - --------------------- - It maps straight onto remediation. A node built only from one subject can - be removed outright. A node built from several cannot - the others still - need it, so it has to be rebuilt without the withdrawn one. - - This function does not assign a contribution class. It reports structure. - Classification needs the class on each edge, which does not exist yet. + Core entry point. `subjects` maps an opaque id to the nodes where its + material enters. Returns per subject: affected (all reachable), + exclusive (reachable from this subject only) and shared. Exclusive nodes + can be removed outright; shared ones must be rebuilt without the + removed input. Structure only, no classes. """ forward = forward_index(graph["edges"]) diff --git a/clew/graph/contribution.py b/clew/graph/contribution.py index 39e7c8a..562c0c2 100644 --- a/clew/graph/contribution.py +++ b/clew/graph/contribution.py @@ -1,52 +1,19 @@ """ -Clew core — contribution class and remediation. - -THIS FILE KNOWS NOTHING ABOUT BIOLOGY. It defines what removability means and -what follows from it. Domain adapters decide which of their events map to -which class; they do not get to invent new ones. - -THE CLASSES ------------ -Given B = f(A1 ... An), remove Ai: - - SEPARABLE There is an efficient g where B' = g(B, Ai). - Invertible IN THE OUTPUT - subtract the contribution - without re-running f. - - REGENERABLE No such g, but f is available and re-runnable, so - B' = f(A1 ... without Ai). - Invertible VIA RE-EXECUTION. - - IRREDUCIBLE Neither. - -The axis is invertibility of a derivation with respect to one input. It is not -ring theory - do not claim an algebraic pedigree for it. - -The enum is CLOSED. If a domain could add a fourth class, core would not know -how to traverse it and determinism would be lost. - -STORAGE MUTABILITY is a second, orthogonal dimension. A contribution can be -SEPARABLE while the artifact is physically unwritable. Whether a remediation -is POSSIBLE and whether it is EXECUTABLE are different questions, and the -answer is the join of the two. - -WHAT LIVES HERE, AND WHAT DOES NOT ----------------------------------- -This file owns the VOCABULARY: the classes, the storage states, the actions, -and the fail-closed normalisation. It does not decide anything. - -Deciding — which combination of dimensions yields which action — lives in -core/policy.py, as a versioned table with a content hash. The split is not -tidiness. The words have to be stable for Clew to mean anything, while the -table has to be versioned so a plan from March can be replayed under the -table that was in force in March. Stable and versioned are different -requirements, so they are different files. - -FAIL CLOSED ------------ -Unknown class becomes IRREDUCIBLE. The two error directions are not symmetric: -over-claiming remediation wastes work, under-claiming tells someone their data -is gone when it is not. Only the second one ends up in front of a regulator. +Contribution classes, storage states and actions: the vocabulary. + +Given B = f(A1 ... An), removing Ai is SEPARABLE when some g gives B' = g(B, +Ai), REGENERABLE when f can be re-run without Ai, and IRREDUCIBLE otherwise. +The axis is invertibility with respect to one input. The enum is closed: a +provider maps its events onto these classes and cannot add one. + +Storage mutability is a second, independent dimension. Whether a remediation +is possible and whether it is executable are different questions. + +An unknown class becomes IRREDUCIBLE. Over-claiming remediation wastes work; +under-claiming tells someone their data is gone when it is not. + +Which combination yields which action belongs to the policy, versioned in +policy.py. The words here must stay stable; the table must be versioned. """ from clew.graph.graph import local_workdir @@ -70,7 +37,7 @@ # --- remediation actions ---------------------------------------------------- PURGE = "PURGE" # remove the contribution, artifact survives -REGENERATE = "REGENERATE" # recompute without the withdrawn source +REGENERATE = "REGENERATE" # recompute without the removed source QUARANTINE = "QUARANTINE" # cannot remediate; block further use DESTROY = "DESTROY" # the artifact exists only because of this # subject; remove it entirely @@ -105,27 +72,11 @@ def explain(action): def storage_state(task, work_root=None): """ - Whether the task's artifacts are still on disk — or None for "not checked". - - NONE IS NOT A THIRD OUTCOME, IT IS THE ABSENCE OF ONE. Storage is a live - property of the world, and the person asking Clew a question is often not - standing where the pipeline ran: a different host, a CI runner, a laptop - reading a graph someone emailed them. Guessing there is not conservative - in either direction, so this refuses. - - In particular DESTROYED is now only ever returned after actually looking - and not finding. It used to be returned whenever `is_dir()` was false, - which fired identically when the path was never recorded, when the volume - was not mounted, when the graph came from another machine, and when the - fixtures were anonymised for publication. All of those became - ALREADY_GONE — "no longer exists; nothing to do" — which is the one error - direction this project exists not to make. A false negative that silences - an obligation is worth more care than a false positive that wastes work. - - `work_root` is the caller saying where to look. Where under it the task - ran is the graph's `workpath`, or for older hashed-layout graphs the last two - components of the recorded path; see graph.local_workdir. A task whose - directory cannot be placed under the root is not checked, not gone. + Whether the task's artifacts are on disk, or None for not checked. None + is the absence of an answer: DESTROYED is returned only after looking + under `work_root` and not finding, never because a path was unrecorded, + unmounted or from another machine. Those used to read ALREADY_GONE, the + one error this project must not make. See graph.local_workdir. """ return storage_at(local_workdir(task, work_root)) @@ -141,14 +92,10 @@ def classify(graph, task_hash, exclusive, published=None, work_root=None, resolved=None): """ Contribution class and storage for one affected task, from pipeline - evidence alone: a task whose script and container were recorded can be - re-executed (REGENERABLE); one without fails closed to IRREDUCIBLE. - Publication arrives as an external assertion and sets `terminal`. - - `storage` is None unless `work_root` says where to look. A caller that - has already placed every task with graph.resolve_workdirs passes the - map as `resolved`, so its refusals (shared directories, a root nothing - exists under) hold here too. See storage_state. + evidence: a recorded script and container mean REGENERABLE, otherwise + IRREDUCIBLE. Publication arrives as an assertion and sets released. + storage is None unless work_root says where to look; pass `resolved` + from graph.resolve_workdirs to keep its refusals. """ task = graph["tasks"].get(task_hash, {}) if resolved is not None: @@ -183,7 +130,7 @@ def classify(graph, task_hash, exclusive, published=None, work_root=None, return { "contribution": klass, "storage": storage, - "exclusive": exclusive, - "terminal": assertion is not None, - "reason": reason, + "scope": "exclusive" if exclusive else "shared", + "released": assertion is not None, + "evidence": reason, } diff --git a/clew/graph/graph.py b/clew/graph/graph.py index 184fb3a..30ccb2d 100644 --- a/clew/graph/graph.py +++ b/clew/graph/graph.py @@ -1,5 +1,5 @@ """ -Clew core — questions you can ask any graph. +Clew core, questions you can ask any graph. Everything here reads the common schema and nothing else: `tasks`, `edges`, `outputs`. No engine, no domain. @@ -27,14 +27,11 @@ def parse_image(image): """ - (name, components, version) of a container reference, version None - when the reference carries none. - - Forms read: registry/path/name:tag, name@sha256:..., Wave images - (tools joined by `_`, hash tag), Singularity cache names (`:` and `/` - turned to `-`, `.img` or `.sif`), conda@hash, and a bare name-version. - Components are the pieces a needle can name: the whole name plus its - `_` and `-` parts, so "samtools" finds "bwa_htslib_samtools". + (name, components, version) of a container reference; version None when + absent. Reads registry paths, name@sha256, Wave images (tools joined by + _), Singularity cache names, conda@hash and a bare name-version. + Components are the name and its _ and - parts, so "samtools" finds + "bwa_htslib_samtools". """ text = (image or "").strip() for scheme in ("docker://", "oras://", "library://", "shub://"): @@ -139,13 +136,13 @@ def external_input_entry_nodes(graph, filename): def load_assertions(path): """ Externally-asserted facts the pipeline cannot know about itself - (publication, so far). Returns {task_hash: assertion_record}; a missing - path honestly means "publication status unknown". + (release, so far). Returns {task_hash: assertion_record}; a missing + path honestly means "release status unknown". """ if not path: return {} data = json.loads(Path(path).read_text()) - return {rec["task"]: rec for rec in data.get("published", [])} + return {rec["task"]: rec for rec in data.get("released", [])} def outputs_for(graph, task_hashes): @@ -190,15 +187,11 @@ def relative_to(path, root): def local_workdir(task, work_root): """ - The task directory under work_root, or None when the graph does not - say where under the root the task ran. - - `workpath` is the extractor's answer, relative to the engine's root, so - a graph from another host still resolves. Without it the recorded - absolute path is only trusted when it visibly follows the hashed - layout; any other shape (one directory shared by every task, a nested - call tree) would resolve to somewhere wrong and read as DESTROYED or, - worse, as deletable. + The task directory under work_root, or None when the graph cannot place + it. `workpath` is the extractor's answer relative to the engine root. + Without it the recorded absolute path is trusted only when it follows + the hashed layout; any other shape would resolve wrongly and read as + DESTROYED or deletable. """ if not work_root: return None @@ -213,13 +206,9 @@ def local_workdir(task, work_root): def resolve_workdirs(graph, work_root): """ {task hash: local directory or None} for every task, plus warnings. - - Two refusals, both in the direction of not claiming anything: tasks - that resolve to one shared directory are all unresolved, because a - verdict on the directory would be a verdict on every one of them; and - when no resolved directory exists at all the root is more likely wrong - than the whole run gone, so every task is unresolved rather than - DESTROYED. + Tasks resolving to one shared directory are all unresolved, and when + nothing resolves at all the root is taken as wrong rather than the run + as gone. """ resolved = {h: local_workdir(t, work_root) for h, t in graph["tasks"].items()} warnings = [] @@ -311,6 +300,9 @@ def task_status(engine_word): # for a call tree), so --work-root plus workpath is the directory on this # machine. Absent when the engine has no per-task directory. OPTIONAL_TASK_FIELDS = ("task_id", "target", "workpath") +# `metrics`: {name: non-negative number}, whatever the engine recorded +# about what a task cost. The names are the provider's; a plan sums each +# per verdict and says how many tasks carried no figure under that name. EDGE_FIELDS = ("consumer", "producer", "filename", "target") EXTERNAL = "EXTERNAL" @@ -338,6 +330,12 @@ def contract_violations(graph): for field in OPTIONAL_TASK_FIELDS: if task.get(field) is not None and not isinstance(task[field], (str, int)): problems.append(f"task {key}: {field} is not a string or integer") + metrics = task.get("metrics") + if metrics is not None and not ( + isinstance(metrics, dict) + and all(isinstance(k, str) and isinstance(v, (int, float)) + and not isinstance(v, bool) and v >= 0 for k, v in metrics.items())): + problems.append(f"task {key}: metrics must map names to non-negative numbers") workpath = task.get("workpath") if isinstance(workpath, str) and ( workpath.startswith("/") or ".." in Path(workpath).parts): diff --git a/clew/graph/results.py b/clew/graph/results.py new file mode 100644 index 0000000..4177b25 --- /dev/null +++ b/clew/graph/results.py @@ -0,0 +1,49 @@ +"""The published results tree: which of a task's outputs were copied there.""" + +from pathlib import Path + +# Outputs every task writes for bookkeeping and no consumer reads, so no +# edge names them and a missing copy is not a missing result. +BOOKKEEPING = ("versions.yml",) + +# Index key standing in for a directory's size, which is meaningless. +DIRECTORY = "directory" + + +def index_results(results_dir): + """ + {(basename, size): [relative paths]}, plus {(basename, DIRECTORY): [...]}. + + Published files are copies, and engine checksums change on copy, so + name plus exact size is the identity. Collisions list every candidate + and are flagged ambiguous rather than picked. + """ + index = {} + root = Path(results_dir) + for path in root.rglob("*"): + rel = str(path.relative_to(root)) + if path.is_file(): + index.setdefault((path.name, path.stat().st_size), []).append(rel) + elif path.is_dir(): + index.setdefault((path.name, DIRECTORY), []).append(rel) + return index + + +def published_copies(graph, task_hash, results_index): + """Published copies of one task's outputs, from the (name, size) index.""" + if not results_index: + return [] + matches = [] + for detail in graph.get("output_details", {}).get(task_hash, []): + name = Path(detail["file"]).name + found = results_index.get((name, detail.get("size")), []) + if found: + matches.append({"output": detail["file"], "published": sorted(found), + "ambiguous": len(found) > 1}) + continue + found = results_index.get((name, DIRECTORY), []) + if found: + # Name only; a directory's size says nothing. Over-report. + matches.append({"output": detail["file"], "published": sorted(found), + "ambiguous": True, "match": "directory name"}) + return matches diff --git a/clew/graph/triggers.py b/clew/graph/triggers.py index c1692e6..10b4e9e 100644 --- a/clew/graph/triggers.py +++ b/clew/graph/triggers.py @@ -1,20 +1,12 @@ """ -Clew core — locating where something bad enters a graph. - -A trigger names what went wrong and resolves to the nodes it entered at. -Every kind answers the same question, and they differ only in where they -look: - - container:toolkit-2.1 a node field - script:prep.py a node field - input:reference.dat an EXTERNAL edge's filename - subject:batch_017 a node's label - site:north also a node's label - -Anything not a known kind is read as a label key, so a graph carrying -`labels: {site: north}` answers `site:north` without a line of code -being added here. That is the point: a vocabulary travels inside the -graph rather than being compiled into the tool. +Where something bad enters a graph. + +A trigger names what went wrong and resolves to the nodes it entered at: +container:toolkit-2.1 and script:prep.py match a node field, +input:reference.dat an EXTERNAL edge's filename, subject:batch_017 a label. +Any other kind is read as a label key, so a graph carrying labels: {site: +north} answers site:north with no code added here. The vocabulary travels +inside the graph. """ from clew.graph.graph import container_matches, external_input_entry_nodes @@ -42,6 +34,16 @@ def external_filename(graph, value): return external_input_entry_nodes(graph, value)[f"input:{value}"] +def label_keys(graph): + """Every label key any task or edge carries.""" + keys = set() + for task in graph["tasks"].values(): + keys.update(task.get("labels") or {}) + for edge in graph["edges"]: + keys.update(edge.get("labels") or {}) + return keys + + def label(key): """ Nodes carrying `labels[key] == value`, on the node or on any artifact diff --git a/clew/ledger/bundle.py b/clew/ledger/bundle.py index be3dc8b..3043c68 100644 --- a/clew/ledger/bundle.py +++ b/clew/ledger/bundle.py @@ -1,62 +1,19 @@ """ -Clew core — the evidence bundle. - -THIS FILE KNOWS NOTHING ABOUT BIOLOGY. It packages opaque documents, hashes -them, and re-checks them. - -WHAT A BUNDLE IS FOR --------------------- -Clew claims three things. A bundle is the artifact that lets someone else -check all three without trusting us, without our database, and without our -code being the thing that says so: - - 1. the log is append-only and unmodified -> the bundled entries re-chain - 2. the computation is deterministic -> the bundled policy is the one - the plan cites, by hash - 3. the result follows from the inputs -> every verdict is RECOMPUTED - from the bundled facts - -The third check is the one that matters and the one that is usually missing -from things called evidence packages. A folder of documents proves only that -somebody assembled a folder. Re-deriving each verdict from the facts and the -table, offline, is what makes the plan a conclusion rather than an assertion. - -NO CLOCK IN THE BUNDLE ----------------------- -Building the same bundle from the same inputs produces the same bundle hash. -That is deliberate and it is testable. A timestamp inside would change the -hash on every build and quietly destroy the reproducibility claim. - -Time is not lost, it is just kept where it belongs: the bundle anchors to a -log head, and the log is the thing with clocks. Sealing a bundle is itself an -event, so "when" is answered by the log, dated and hash-chained, rather than -by a field the builder could have typed anything into. - -HOW THIS CLOSES THE LOG'S OPEN GAP ----------------------------------- -A hash chain detects editing but not truncation: lopping entries off the end -leaves a shorter, self-consistent chain, and a rewrite by whoever holds the -owner's credentials leaves no trace at all. Nothing inside the database can -fix that — the fix has to be a witness the database's owner does not control. - -A bundle is that witness. It records the log head it covered, and it goes out -of the building: to an assessor, into a build artifact, to a partner. Bundles -also chain to each other, so a sequence of them pins a sequence of heads. To -make a truncation stick, someone would now have to collect every copy of -every bundle ever issued. - -WHAT "SIGNED" HONESTLY MEANS HERE ---------------------------------- -This module SEALS: a SHA-256 manifest over every file, and a bundle hash over -the manifest. That needs nothing but the standard library, so anyone can -verify it, which is the whole point. - -It does NOT implement signing. A signature that can only be checked by -someone holding the signing key is not a signature in the sense an assessor -means, and inventing crypto here would be indefensible. Countersigning is -detached and delegated to tooling the reader already trusts and already -manages keys for — ssh-keygen -Y, which ships with OpenSSH. The seal is -Clew's; the attestation of WHO sealed it belongs to your key infrastructure. +The evidence bundle: opaque documents, hashed, checkable offline. + +Three checks, one per claim. The bundled log entries re-chain. The bundled +policy is the one the plan cites, by hash. Every verdict is recomputed from +the bundled facts, which makes the plan a conclusion rather than a folder +somebody assembled. + +No clock inside, so the same inputs give the same bundle hash. Time lives in +the log the bundle anchors to, and sealing is itself a logged event. The +bundle records the log head it saw and leaves the building, so a truncated +log is caught by a witness its owner does not control. + +This module seals with SHA-256 and the standard library. It does not sign. +Countersigning is delegated to ssh-keygen -Y, which readers already trust +and manage keys for. """ import hashlib @@ -118,12 +75,9 @@ def bundle_hash(manifest): def _crate(documents, description): """ - A minimal RO-Crate 1.1 description of the bundle. - - Adopted rather than invented: labs already publish crates for journals and - archives, and Clew already ingests them. A bundle that is also a crate is - one fewer format for a reader to learn, and it survives being handed to - tooling that knows nothing about Clew. + A minimal RO-Crate 1.1 description of the bundle. Labs already publish + crates and Clew already reads them, so a bundle that is also a crate is + one fewer format. """ parts = sorted(set(documents) | {MANIFEST}) return { @@ -192,23 +146,12 @@ def build(destination, documents, log_head, previous_bundle=None, previous_log_head=None, since=0, coverage=None, description="Clew evidence bundle", force=False): """ - Write a bundle and return its manifest and hash. - - `documents` maps a filename to a JSON-serialisable object. Core does not - know or care what any of them mean; the caller decides what belongs. - - `log_head` is {seq, hash} for the log this bundle witnesses. `since` is - the seq the bundled entries start after; the hash they must chain back - to is genesis when that is 0 and otherwise the head of the previous - bundle, so `previous_log_head` ({seq, hash}) and `previous_bundle` (its - hash) are required for a window. A sequence of bundles then pins a - sequence of log heads, and each window is verifiable against the one - before it rather than against itself. - - A destination that already holds files is refused unless `force`, - which empties it first. Building into a directory with leftovers would - seal whatever happened to be there, or fail to verify for a reason the - builder never saw. + Write a bundle and return its manifest and hash. `documents` maps a + filename to a JSON-serialisable object; core does not interpret them. + `log_head` is {seq, hash}. `since` is the seq the entries start after, + chaining to genesis at 0 and otherwise to `previous_log_head` and + `previous_bundle`, so each window verifies against the one before. A + non-empty destination is refused unless `force`. """ for name in documents: if not safe_name(name): @@ -326,17 +269,11 @@ def chain_start(manifest): def verify_log(events, manifest, eventlog): """ - The bundled entries must re-chain from the recorded start and end at the - recorded head. - - The start is not taken from the entries themselves. A chain checked - against its own first prev_hash verifies whatever it was forged to - say; it is checked against genesis, or against the head of the bundle - it continues, which the manifest names. - - `eventlog` is passed in rather than imported so this stays usable in an - environment with no database driver installed — which is exactly the - environment an auditor checking a bundle is in. + The bundled entries must re-chain from the recorded start to the + recorded head. The start comes from the manifest, genesis or the + previous bundle's head, never from the entries themselves, which would + verify whatever they were forged to say. `eventlog` is passed in so this + runs without a database driver. """ anchors = manifest["anchors"] head = anchors["log_head"] @@ -394,16 +331,10 @@ def verify_log(events, manifest, eventlog): def verify_against_log(manifest, hash_at_seq): """ - Hold a live log up against what this bundle witnessed. - - `hash_at_seq` is a callable taking a sequence number and returning that - entry's hash, or None if the log has no such entry. A callable rather - than a connection so this stays driver-free and testable. - - This is the check that closes the log's open gap. A truncated chain is - internally consistent and verify() on the log alone passes — nothing - inside the database can notice something that is no longer in it. A - bundle can, because it left the building carrying the head it saw. + Hold a live log against what this bundle witnessed. `hash_at_seq` + returns an entry's hash by sequence number, or None. A truncated chain + passes verify() on its own; a bundle that left the building carrying the + head it saw catches it. """ anchor = manifest["anchors"]["log_head"] if anchor["seq"] == 0: @@ -489,12 +420,9 @@ def verify_replay(plan, policy_document): for item in items: try: decision = policy_module.decide( - item["contribution"], - storage=item.get("storage"), - exclusive=item.get("exclusive"), - terminal=item.get("terminal"), - policy=policy_document, - ) + item["contribution"], storage=item.get("storage"), + scope=item.get("scope"), released=item.get("released"), + mode=item.get("mode"), policy=policy_document) except ValueError as exc: mismatches.append(f"{item['task']}: {exc}") continue diff --git a/clew/ledger/eventlog.py b/clew/ledger/eventlog.py index 11ee6dd..ec5f2b5 100644 --- a/clew/ledger/eventlog.py +++ b/clew/ledger/eventlog.py @@ -1,77 +1,22 @@ """ -Clew core — the append-only event log, on Postgres. - -THIS FILE KNOWS NOTHING ABOUT BIOLOGY. It stores opaque events: a type, a -subject, a body, an actor, two timestamps. Core never interprets any of them. -Domain adapters define what the types mean; if core ever needed to know, the -boundary would be broken. - -WHAT THIS IS FOR ----------------- -Clew claims three things and only three. This file is the first of them: - - the log is append-only and unmodified. - -Everything downstream — a blast radius, a remediation plan, an evidence -bundle — is a computation over facts. If the facts can be edited after the -fact, none of the rest is worth anything. So the facts get their own store -whose only job is to make revision impossible for the application and -detectable for everyone else. - -TWO CLOCKS, DELIBERATELY ------------------------- - effective_from when the fact became true in the world - recorded_at when we learned it - -They are usually different and the gap is the interesting part. A withdrawal -signed on the 1st and entered on the 5th was true from the 1st; a release -made on the 3rd was made in good faith and is still a release that must be -disclosed. One timestamp cannot express that, and retrofitting the second one -later means re-interpreting every historical row. So both, from the start. - -Note what this does NOT do: it does not put a clock in the computation. -The log timestamps facts. A plan computed from those facts is a pure function -of them, and stays byte-identical on replay. Learning is dated; deciding is not. - -THREE LAYERS OF PROTECTION, EACH WITH A DIFFERENT JOB ------------------------------------------------------ -They are not redundant. Each one stops something the next cannot. - -1. ROLE GRANTS stop the application. - The writer role holds SELECT and INSERT. It was never granted UPDATE, - DELETE or TRUNCATE, and it cannot grant them to itself. This is the reason - this log is on a server rather than in a local file: enforcement lives - outside the process that writes, in a database the application does not - administer. A file-backed store cannot do this — whoever holds the file - holds everything. - -2. TRIGGERS stop the owner's mistake. - Grants do not constrain the table owner, and the owner is a real person - with a psql prompt at 6pm. The triggers refuse UPDATE, DELETE and TRUNCATE - from anyone, owner included. TRUNCATE gets its own statement-level trigger - because it does not fire row triggers at all — it would otherwise empty the - whole log silently. - -3. THE HASH CHAIN catches whoever defeats both. - A superuser can disable a trigger and rewrite a row. Every entry hashes its - own content plus its predecessor's hash, so any such edit fails to - recompute, and verify() finds it without trusting the table, the triggers, - or us. - -WHAT IS STILL NOT PROVED ------------------------- -The chain detects EDITING. It does not detect TRUNCATION OF THE TAIL — a -shorter chain is still self-consistent — nor a FULL REWRITE by someone who -rebuilds every hash from the altered point. - -No hash chain solves that alone. What closes it is anchoring the head hash -somewhere the log's owner does not control: a countersigned evidence bundle, a -build log, a timestamping service. Slice 3 does that. Until then the claim is -exactly "the application cannot edit, and anyone else's edit is detectable", -which is what this says rather than something more comfortable. - -The tests assert both limits rather than leaving them implied, so nobody -later reads verify()'s ok=True as "nothing was lost". +The append-only event log, on Postgres. It stores opaque events: type, +subject, body, actor and two timestamps. Providers define what the types +mean. + +Two clocks. effective_from is when the fact became true; recorded_at is when +the log heard it. The gap is often the interesting part, and a plan computed +from the facts carries no clock of its own. + +Three protections with different jobs. Role grants stop the application: the +writer holds SELECT and INSERT and cannot grant itself more, which no file +can offer. Triggers stop the owner's mistake by refusing UPDATE, DELETE and +TRUNCATE from anyone. The hash chain catches whoever defeats both: each +entry hashes its content and its predecessor, so an edit fails to recompute +and verify() finds it without trusting the table. + +Not proved here: truncation of the tail, or a rewrite from a point onward. A +witness outside the owner's control closes that, and the evidence bundle is +that witness. """ import hashlib @@ -120,7 +65,7 @@ def event_hash(entry): fields = {k: entry[k] for k in FIELDS if k != "hash"} # The body is hashed as its canonical TEXT, which is what the database # stores. Accepting a structure here and canonicalising it means an - # exported bundle verifies whether its bodies arrive parsed or raw — + # exported bundle verifies whether its bodies arrive parsed or raw , # a trap worth closing once rather than in every consumer. if not isinstance(fields["body"], str): fields["body"] = canonical(fields["body"]) @@ -130,23 +75,11 @@ def event_hash(entry): def verify_entries(entries, start_seq=1, start_prev=GENESIS): """ - Re-walk a chain in sequence order, recomputing every hash. - - Takes plain dicts, not rows, so the same function verifies a live log and - an exported evidence bundle. Trusts nothing but the raw field values: - not the stored hash, not the triggers, not that the file opened cleanly. - Anyone can run this, which is the property that matters — an auditor who - does not trust us can verify without us. - - `start_seq`/`start_prev` anchor a WINDOW of the chain. A bundle covering - entries 40-60 is not verifiable on its own — it is verifiable against the - hash the previous bundle ended on. Defaulting them to the genesis pair - means an unanchored window silently verifies as if it were the whole log, - so a caller checking a window must supply the anchor it is claiming. - - Returns the FIRST failure, not a list. A chain is broken from its first - bad link onwards; reporting every later entry as "also wrong" would be - noise that buries where the edit actually happened. + Re-walk a chain in sequence order, recomputing every hash from the raw + field values. Plain dicts, so a live log and a bundle verify the same + way. `start_seq` and `start_prev` anchor a window; an unanchored window + would pass as if it were the whole log. Returns the first failure only, + which is where the edit happened. """ expected_prev = start_prev expected_seq = start_seq @@ -193,17 +126,10 @@ def now(): def instant(text): """ - An ISO-8601 string as an aware UTC datetime, for comparing and sorting. - - Timestamps are stored as text because the hash covers bytes, but text - compares as text: "2026-03-01T09:00:00+02:00" sorts after - "2026-03-01T08:00:00+00:00" although it is the earlier instant, and a - date-only value never compares equal to the same day with a time on it. - Every comparison of two timestamps goes through here instead. - - A date alone means midnight UTC. A time with no offset is taken as UTC. - Anything unparseable raises, because a comparison against a value that - cannot be placed on a timeline has no honest answer. + An ISO-8601 string as an aware UTC datetime. Timestamps are stored as + text because the hash covers bytes, but text order is not time order + across offsets, so every comparison goes through here. A bare date is + midnight UTC, a missing offset is UTC, and an unparseable value raises. """ if not isinstance(text, str) or not text.strip(): raise ValueError(f"not a timestamp: {text!r}") @@ -265,7 +191,7 @@ def in_effect(effective_from, as_of): -- Timestamps are text, not timestamptz, and that is deliberate. The hash -- covers bytes. A timestamptz round-trips through the server's own -- formatting, so '+00:00' could come back as 'Z' and every hash after it --- would fail to recompute — verification broken by a display convention. +-- would fail to recompute, verification broken by a display convention. -- ISO-8601 UTC sorts correctly as text, which is the whole reason the format -- exists, so nothing is lost for querying. @@ -308,21 +234,12 @@ def connect(dsn, autocommit=True): def init(conn, writer=WRITER_ROLE, auditor=AUDITOR_ROLE, writer_password=None, auditor_password=None): """ - Create the table, the guards, and the two roles. Run as the owner. - - THE GRANTS ARE THE POINT. The writer is given SELECT and INSERT and - nothing else — not UPDATE, not DELETE, not TRUNCATE — and a role cannot - grant itself a privilege it does not hold. The application therefore - cannot edit the log even if its code is wrong, its credentials leak, or - someone is in a hurry. That is prevention, and it is the one thing a - file-backed store cannot offer at any price: whoever holds the file holds - every privilege over it. - - What this does NOT constrain is the owner, who can drop a trigger and - re-grant anything. Owner and application must therefore be different - identities, and the owner's credentials should not live in the pipeline. - Clew cannot enforce that from inside; it is an operational control, and - saying so is more use than implying we solved it. + Create the table, the guards and the two roles, as the owner. The writer + gets SELECT and INSERT only and cannot grant itself more, so the + application cannot edit the log whatever its code or credentials do. The + owner can still drop a trigger, so owner and application must be + different identities. That is an operational control, not something this + code enforces. """ from psycopg import sql @@ -373,22 +290,11 @@ def head(conn): def append(conn, event_type, subject, body=None, actor="unknown", effective_from=None): """ - Add one event and return it, hash included. - - `recorded_at` is the server's clock, read inside the same transaction - that writes the row. The caller cannot supply it: recorded_at is the - one field whose value is "when the log heard this", and the process - doing the telling is the wrong party to say when that was. The hash - covers it, so it cannot be stamped inside the INSERT itself; reading - the server clock first and hashing over that is the same guarantee. - - `effective_from` defaults to `recorded_at` — a fact with no stated - effective date is treated as effective when we heard it. That default - never back-dates anything on its own, which is the safe direction, but it - is still a default: a domain that knows the real date should pass it. - A value that is not an ISO-8601 timestamp is refused here, because a - fact that cannot be placed on a timeline cannot be ordered against any - other and would poison every later comparison. + Add one event and return it, hash included. `recorded_at` is the server + clock read in the same transaction; the caller is the wrong party to say + when the log heard it. `effective_from` defaults to recorded_at, which + never back-dates, and must be ISO-8601 or it cannot be ordered against + anything. """ if effective_from is not None: instant(effective_from) @@ -460,7 +366,7 @@ def read(conn, since=0, until=None, event_type=None, subject=None): def raw(conn, since=0, until=None): """ - Entries with bodies left as stored text — the form that was hashed. + Entries with bodies left as stored text, the form that was hashed. verify() and the evidence bundle both want this. read() is for humans and for code that wants structures; raw() is for arithmetic. @@ -479,12 +385,9 @@ def raw(conn, since=0, until=None): def anchor(conn, seq): """ - The hash an entry range beginning after `seq` must chain back to. - - Genesis when seq is 0. Raises when the named entry does not exist, rather - than falling back to genesis: a missing anchor means the caller is - verifying against a history that is not there, and quietly treating that - as "start of log" would turn a real problem into a pass. + The hash an entry range beginning after `seq` must chain to: genesis at + 0. A missing entry raises rather than falling back to genesis, which + would turn a check against absent history into a pass. """ if seq == 0: return GENESIS diff --git a/clew/ledger/evidence.py b/clew/ledger/evidence.py index 1b09c4b..5c5ab69 100644 --- a/clew/ledger/evidence.py +++ b/clew/ledger/evidence.py @@ -1,37 +1,26 @@ """ -Clew — build and check evidence bundles. +Build and check evidence bundles. - # seal a plan, the policy it used, and the log entries behind it - clew evidence build --out bundle/ --plan plan.json \ - --dsn "$CLEW_DSN" --input graph.json --input donors.csv - - # anyone, anywhere, with Python and nothing else + clew evidence build --plan plan.json --dsn "$CLEW_DSN" \ + --input graph.json # -> container-build-1.2.3-2026-09-15/ + clew evidence build --out bundle/ --plan plan.json clew evidence verify bundle/ - - # countersigning is delegated, not invented clew evidence sign bundle/ --key ~/.ssh/id_ed25519 clew evidence verify bundle/ --allowed-signers allowed_signers -WHY THE VERIFIER TAKES NO CREDENTIALS -------------------------------------- -`verify` reads a directory. It does not connect to the database, does not -call us, and does not need the driver installed. An assessor who does not -trust the party that produced a bundle must be able to check it anyway, and -any step that routes through the producer's infrastructure defeats that. - -WHY BUILDING IS SEPARATE FROM COMPUTING ---------------------------------------- -clew impact answers a question. This packages an answer that was already given. -Keeping them apart means a bundle can only ever contain a plan that was -produced independently — sealing cannot quietly recompute something on more -convenient terms on its way into the folder. +verify reads a directory: no database, no driver, no call home, so an +assessor who does not trust the producer can still check. Building is +separate from computing so a bundle can only hold a plan produced +independently; sealing cannot recompute on more convenient terms. """ import argparse import json import os +import re import shutil import subprocess +from datetime import datetime, timezone import sys from pathlib import Path @@ -60,7 +49,7 @@ def open_log(dsn): def resolve_policy_for(plan, override): """ - The exact table the plan was decided under — never a substitute. + The exact table the plan was decided under, never a substitute. A plan citing a version this build does not ship cannot be sealed here. Quietly bundling today's table instead would produce a bundle whose replay @@ -70,12 +59,13 @@ def resolve_policy_for(plan, override): document = policy_module.resolve_or_load(override) else: version = plan.get("policy_version") - if version not in policy_module.REGISTRY: + known = policy_module.available() + if version not in known: raise SystemExit( f"the plan cites policy {version!r}, which this build does not " - f"ship ({', '.join(sorted(policy_module.REGISTRY))}). Pass " + f"know ({', '.join(sorted(known))}). Pass " "--policy pointing at the table it was computed under.") - document = policy_module.resolve(version) + document = known[version] stated = plan.get("policy_hash") actual = policy_module.fingerprint(document) @@ -88,12 +78,8 @@ def resolve_policy_for(plan, override): def record_inputs(paths): """ - Source files by content hash, with the basename only. - - Keyed by hash rather than by path so the same inputs give the same - bundle whatever directory they were read from: a bundle hash that - changed because someone typed ./graph.json instead of graph.json - would be a reproducibility claim with a hole in it. + Source files by content hash with basename only, so the same inputs give + the same bundle from any directory. """ inputs = {} for path in paths or []: @@ -108,6 +94,17 @@ def record_inputs(paths): return inputs +def default_out(plan, today=None): + """ + -, so the name says what the bundle answers. The date is + in the name only; nothing sealed depends on the clock. + """ + trigger = str(plan.get("trigger") or "plan") + slug = re.sub(r"[^a-z0-9._]+", "-", trigger.lower()).strip("-.") or "plan" + day = today or datetime.now(timezone.utc).date().isoformat() + return f"{slug}-{day}" + + def continue_from(previous_dir, since): """The previous bundle's hash and log head, checked against --since.""" if not previous_dir: @@ -132,6 +129,7 @@ def continue_from(previous_dir, since): def cmd_build(args): plan = load_json(args.plan) + args.out = args.out or default_out(plan) policy_document = resolve_policy_for(plan, args.policy) previous, previous_head = continue_from(args.previous, args.since) @@ -319,12 +317,9 @@ def cmd_verify(args): def cmd_witness(args): """ - Hold a live log up against a bundle that has already left the building. - - Deliberately a separate command from `verify`. Verifying a bundle needs - no credentials and must stay that way; this one needs the log, and - folding it into `verify` would make the credential-free property look - optional when it is the whole design. + Hold a live log against a bundle that has left the building. Separate + from verify on purpose: verify needs no credentials, and folding this in + would make that look optional. """ from clew.ledger import eventlog @@ -363,7 +358,7 @@ def check_signature(directory, allowed_signers): # Ask the signature who signed it, then check that claim against the # allowed_signers file. Verifying requires naming a principal, and an - # auditor holding a bundle has no reason to know one in advance — the + # auditor holding a bundle has no reason to know one in advance, the # useful output here is WHO attested to it, which this recovers. found = subprocess.run( ["ssh-keygen", "-Y", "find-principals", "-s", str(signature), @@ -393,12 +388,9 @@ def check_signature(directory, allowed_signers): def cmd_sign(args): """ - Countersign the manifest with ssh-keygen. Clew implements no crypto. - - The manifest is the right thing to sign: it already covers every file by - hash, so one signature over it attests to the whole bundle, and the - signature can be added, replaced or made by several parties without any - of the sealed content changing. + Countersign the manifest with ssh-keygen; Clew implements no crypto. The + manifest covers every file by hash, so one signature attests to the + bundle and can be added or replaced without touching sealed content. """ if not shutil.which("ssh-keygen"): raise SystemExit("ssh-keygen not found; it ships with OpenSSH") @@ -423,7 +415,8 @@ def main(argv=None): sub = parser.add_subparsers(dest="command", required=True) build = sub.add_parser("build", help="seal a plan into a bundle") - build.add_argument("--out", required=True, metavar="DIR") + build.add_argument("--out", metavar="DIR", + help="bundle directory; default -") build.add_argument("--plan", required=True, help="plan JSON from clew impact --json") build.add_argument("--policy", metavar="VERSION|PATH", diff --git a/clew/ledger/gate.py b/clew/ledger/gate.py index fa85434..c2f6f0d 100644 --- a/clew/ledger/gate.py +++ b/clew/ledger/gate.py @@ -1,50 +1,16 @@ """ -Clew core — the pre-flight gate. - -THIS FILE KNOWS NOTHING ABOUT BIOLOGY. It takes a set of opaque subject ids, -a set of opaque facts about them, and a statement of which fact types block. - -WHAT A GATE IS, AND WHY IT IS NOT A BLAST RADIUS ------------------------------------------------- -Everything else in Clew answers a question after the fact: this went wrong, -what must happen now. A gate asks the opposite question, before anything runs: -is any of this material something we are not allowed to use? - -The two are not variations on each other. A blast radius traverses a graph of -work that already happened. A gate looks at a list of inputs and a log of -facts, and there is no graph yet because nothing has run. - -THREE OUTCOMES, NOT TWO ------------------------ - BLOCKED a fact in effect says do not use this - CLEARED a fact in effect says it is fine - UNKNOWN the log has nothing to say about this subject at all - -The third one is the entire reason this file is careful. A gate that reports -UNKNOWN as "not blocked" passes everything it failed to check, and the day -someone mistypes an identifier — or points at the wrong log, or connects with -a role that cannot read — it goes green while checking nothing. - -That is the same failure this project has now made twice: an absent answer -being reported as a clean one. So UNKNOWN is a distinct outcome, it is -counted, and the caller must decide explicitly what to do with it. - -LATEST EFFECTIVE FACT WINS --------------------------- -Decisions get reversed. A subject withdrawn in March and reinstated in June -is usable in July, and a gate that only ever accumulated prohibitions would -be wrong about that in the direction of refusing legitimate work. - -So for each subject, the LATEST DECISIVE FACT IN EFFECT decides — ordered by -effective_from, which is when the decision was made in the world, not by -recorded_at, which is merely when we heard. Two facts effective on the same -date are broken by log order, because the log is the only tiebreak that -cannot be back-dated by whoever entered them. - -Facts effective in the FUTURE relative to the question do not count. Asking -"was this usable on 1 March" must not be answered with a decision taken in -June, or every historical gate result becomes unreproducible the moment -someone records something new. +The admission decision: opaque subject ids, opaque facts, and which fact +types block or clear. + +A blast radius traverses work that happened. A gate looks at a list of +inputs before anything runs. Three outcomes: BLOCKED, CLEARED and UNKNOWN, +where the log has nothing to say. UNKNOWN is its own outcome because a gate +that treats it as clear goes green while checking nothing, the day an +identifier is mistyped or the wrong log is named. + +For each subject the latest decisive fact in effect decides, ordered by +effective_from with log order as the tiebreak. Facts effective after the +question's date do not count, so a historical answer stays reproducible. """ from clew.ledger.eventlog import in_effect, instant, now @@ -56,14 +22,9 @@ def decisive_facts(entries, blocking, clearing, as_of=None): """ - The facts that can decide anything, in the order they took effect. - - `blocking` and `clearing` are sets of opaque type names. Core does not - know what any of them mean — a domain decides which of its vocabulary - belongs in which set, and that mapping is the customer's to author. - - Timestamps are compared as instants, never as text: two offsets spell - the same moment differently, and text order is not time order. + The facts that can decide anything, in effective order. `blocking` and + `clearing` are opaque type names the provider chose. Timestamps compare + as instants, never as text. """ decisive = set(blocking) | set(clearing) relevant = [e for e in entries if e["event_type"] in decisive] @@ -113,19 +74,11 @@ def status_by_subject(subjects, entries, blocking, clearing, as_of=None): def decide(subjects, entries, blocking, clearing, as_of=None, unknown_blocks=True): """ - The gate's verdict, with everything needed to defend it. - - `unknown_blocks` defaults True. A gate exists to stop work that should - not proceed, and the caller who cannot say whether a subject is permitted - has not established that it is. Turning it off is a real and sometimes - correct choice — a pilot run against a log that only covers part of an - estate — but it must be a choice someone made, not a default they - inherited without noticing. - - `as_of` defaults to now and the value used is recorded in the result. - A result that said "as of: nothing in particular" could never be - re-derived once the log grew, because nobody would know which facts - were in view when it was decided. + The verdict with what is needed to defend it. `unknown_blocks` defaults + True: a caller who cannot say a subject is permitted has not established + it, and turning that off must be a choice someone made. `as_of` defaults + to now and is recorded, so the result can be re-derived after the log + grows. """ if as_of is None: as_of = now() diff --git a/clew/ledger/logbook.py b/clew/ledger/logbook.py index 23a7dc0..6ae7a10 100644 --- a/clew/ledger/logbook.py +++ b/clew/ledger/logbook.py @@ -1,40 +1,17 @@ """ -Clew — the event log, from the command line. +The event log from the command line. - # once, as the database owner: table, guards, and the two roles - clew log --dsn "$CLEW_ADMIN_DSN" init \ - --writer-password "$W" --auditor-password "$A" - - # thereafter, as the writer role, which holds INSERT and SELECT only + clew log --dsn "$CLEW_ADMIN_DSN" init --writer-password W --auditor-password A clew log --dsn "$CLEW_DSN" append --type ContainerDefectReported \ --subject "gatk4:4.2.1" --actor qa.lead@example.org \ - --effective-from 2026-08-10T00:00:00+00:00 \ - --body '{"defect":"BQSR miscalibration","reference":"JIRA QA-4471"}' - - # by anyone, including someone who does not trust us + --effective-from 2026-08-10T00:00:00+00:00 --body '{"defect": "..."}' clew log --dsn "$CLEW_AUDIT_DSN" verify -THREE DSNs, THREE IDENTITIES ----------------------------- -That is not ceremony. The owner can drop a trigger; the writer cannot edit; -the auditor cannot write. If the pipeline runs as the owner, the separation -that makes this log worth having is gone, and no amount of Clew code can put -it back. Use different credentials, and keep the owner's out of CI. - -WHY A SEPARATE ENTRY POINT FROM clew impact ----------------------------------------- -Recording a fact and computing over it are different acts by different people -at different times. A coordinator enters a withdrawal months before anyone -runs a blast radius against it. Collapsing the two into one command would -imply they happen together, and would quietly invite writing a fact into the -log as a side effect of asking a question about it. Facts get their own door. - -EVENT TYPES ARE NOT VALIDATED HERE, ON PURPOSE ----------------------------------------------- ---type takes any string. Core defines no vocabulary; a domain adapter decides -what its types mean. Rejecting an unknown type here would put a domain's -vocabulary in core's entry point, which is the boundary this project exists -to keep. What the log guarantees is that whatever was written stays written. +Three DSNs, three identities: the owner can drop a trigger, the writer +cannot edit, the auditor cannot write. A pipeline that runs as the owner has +thrown the separation away. Recording a fact is a separate command from +computing over it, done by a different person at a different time. --type +takes any string; the vocabulary is the provider's, not the log's. """ import argparse @@ -53,7 +30,7 @@ def cmd_init(conn, args): auditor_password=args.auditor_password) print(f"database {result['database']}, schema {result['schema']}") print(f" {result['writer']:<16} SELECT, INSERT (cannot UPDATE, DELETE " - f"or TRUNCATE — never granted, and cannot self-grant)") + f"or TRUNCATE, never granted, and cannot self-grant)") print(f" {result['auditor']:<16} SELECT") print() print("Connect the pipeline as the writer. Keep the owner's credentials") @@ -106,7 +83,7 @@ def cmd_verify(conn, args): print("This proves no entry was EDITED. It does not prove none were") print("dropped from the end, or that the whole chain was not rebuilt") print("by someone holding the owner's credentials. Anchor this head") - print("hash outside the database — an evidence bundle, a build log —") + print("hash outside the database, an evidence bundle, a build log , ") print("and those become detectable too.") return 0 print(f"FAILED at seq {result['broken_at']}") diff --git a/clew/ledger/policy.py b/clew/ledger/policy.py index b953045..87008ed 100644 --- a/clew/ledger/policy.py +++ b/clew/ledger/policy.py @@ -1,90 +1,25 @@ """ -Clew core — the remediation policy, versioned. - -THIS FILE KNOWS NOTHING ABOUT BIOLOGY. It maps four opaque dimensions onto one -action and records which rule did it. - -WHY THIS EXISTS ---------------- -Clew's second claim is that the computation is deterministic and reproducible. -A decision table written as an if-ladder cannot support that claim, for one -reason: editing it silently re-interprets every plan ever produced. A plan -from March says QUARANTINE; the code says QUARANTINE today; nobody can tell -whether it said QUARANTINE in March. The history is unfalsifiable, which is -the same as worthless. - -So the table becomes DATA with a version and a content hash, every decision -names the rule that made it, and every plan carries the policy it was -computed under. "Policy v1, rule R5, these hashes, re-run and get the same -answer" is then a checkable sentence rather than a slogan. - -WHAT IS AND IS NOT VERSIONED HERE ---------------------------------- -This is the CORE decision table: given a contribution class, a storage state, -whether the artifact is exclusively owned, and whether it is terminal, what -must happen. It defines what the classes MEAN, so it is ours, not the -customer's. Changing it changes the semantics of every historical plan, which -is exactly why it needs a version. - -The customer's policy is a different object: which of their events map to -which contribution class, what counts as published, what a given tier of -withdrawal is allowed to reach. That lives in domains/ and is not this file. -CLAUDE.md's "the customer authors the policy, we ship templates" is about -that layer. Conflating the two would let a customer redefine SEPARABLE, and -then no two Clew deployments would mean the same thing by the same word. - -RULES ARE MATCH DICTS, FIRST MATCH WINS ---------------------------------------- - {"id": "R3", "when": {"exclusive": True, "storage": "WRITABLE"}, - "action": "DESTROY", "because": "..."} - -An omitted dimension is a wildcard. This is not a rule engine and must not -grow into one: no negation, no arithmetic, no expressions. The entire -semantics is "does every named field equal this value", and the reason is -that an auditor has to be able to read the policy. A condition language rich -enough to be interesting is rich enough to be argued about. - -AN UNVERIFIED DIMENSION IS NOT A VALUE --------------------------------------- -Storage state is not lineage. Lineage says what was derived from what and is -permanently true; storage says whether the bytes are still there and is true -only at the instant you look. Whoever asks Clew a question may or may not be -standing somewhere that can look. - -When a dimension is unverified, decide() is given None for it and evaluates -the policy once per possible value: - - - every value yields the same action -> that action is certain anyway, - and is returned normally - - the values disagree -> no action is returned at all - -The second case is the whole point of the design. Guessing WRITABLE -over-claims an obligation and wastes work; guessing DESTROYED yields -ALREADY_GONE, which reads as "you have nothing to do" and is the one error -this project exists not to make. Returning neither is the only honest answer, -and it names which verdicts are still in play so the reader knows exactly -what verifying the storage would settle. - -No new action is invented for this. The action enum is closed — a new verdict -would change what remediation means — so an undetermined item simply HAS no -action, and carries the candidates instead. - -TWO GUARDS SIT OUTSIDE THE RULE LIST ------------------------------------- - 1. The contribution class is normalised before matching. Anything - unrecognised becomes IRREDUCIBLE first, and no rule can test for an - unrecognised class because validate() rejects such a rule. - 2. Falling off the end of the rules yields QUARANTINE — not an error, and - not a pass. An incomplete policy is cautious rather than permissive. - -BE PRECISE ABOUT WHAT THAT GUARANTEES, THOUGH. It fixes the FACTS, not the -VERDICT. A policy that maps IRREDUCIBLE to PURGE is expressible, and would be -wrong, and Clew will run it. That is not a hole — it is the reason the table -is data. Wrong logic in an if-ladder is invisible in a code review nobody -does; wrong logic in a hashed, versioned, rationale-carrying policy file is -sitting in the open with a rule id on it. There is a test named for this so -that nobody later mistakes it for an oversight and "fixes" it by hardcoding -verdicts back into core. +The remediation policy: a versioned table, not an if-ladder. + +A rule is a match dict, first match wins, an omitted dimension is a +wildcard, and there is no other syntax, so an auditor can read the whole +policy: + + {"id": "R3", "when": {"scope": "exclusive", "storage": "WRITABLE"}, + "action": "DESTROY", "reason": "..."} + +The table carries a version and a content hash, every decision names its +rule, and every plan carries the policy it ran under, so a plan from March +replays under March's table. This table defines what the classes mean, so +Clew owns it. Mapping a site's events onto classes is the provider's job and +lives elsewhere. + +A dimension passed as None is unverified, not a value: decide() runs once +per possible value and returns no action when they disagree, naming the +candidates instead. Unknown classes normalise to IRREDUCIBLE first, and +falling off the end of the rules yields QUARANTINE. Neither guard fixes the +verdict. A policy mapping IRREDUCIBLE to PURGE loads and runs, and is wrong +in the open with a rule id on it, which is why the table is data. """ import hashlib @@ -95,16 +30,22 @@ # The dimensions a rule may test. A rule naming anything else is rejected at # load time rather than silently never matching. -DIMENSIONS = ("contribution", "storage", "exclusive", "terminal") + +DIMENSIONS = ("contribution", "storage", "scope", "released", "mode") +EXCLUSIVE, SHARED = "exclusive", "shared" # scope: made for this subject alone, or not +REMOVE, TRACE = "remove", "trace" # mode: the subject is gone, or changed +MODES = (REMOVE, TRACE) VALID = { "contribution": set(contribution.CLASSES), "storage": set(contribution.STORAGE), - "exclusive": {True, False}, - "terminal": {True, False}, + "scope": {EXCLUSIVE, SHARED}, + "released": {True, False}, + "mode": set(MODES), } -TYPES = {"storage": str, "exclusive": bool, "terminal": bool} +TYPES = {"storage": str, "scope": str, "released": bool, "mode": str} +VERIFIABLE = ("storage", "scope", "released", "mode") ACTIONS = { contribution.PURGE, contribution.REGENERATE, contribution.QUARANTINE, @@ -123,146 +64,73 @@ FALLTHROUGH_RULE = "fallthrough" -def rule(rule_id, action, because, **when): - return {"id": rule_id, "when": when, "action": action, "because": because} +def rule(rule_id, action, reason, **when): + return {"id": rule_id, "when": when, "action": action, "reason": reason} # --------------------------------------------------------------- the policy +# +# First match wins, so order is part of the rule. Release is asked before +# existence: deleting our copy does not reach the released one. A +# corrected subject's separable part is recomputed, not merely removed. V1 = { "version": "v1", - "description": "Clew's built-in remediation table.", + "description": "Clew's remediation table.", "rules": [ - rule("R1", contribution.ALREADY_GONE, - "Nothing survives to remediate. Asked first because every later " - "question presumes an artifact still exists.", + rule("R1", contribution.NOTIFY_ONLY, + "Released: published, or past a trust boundary. Destroying our " + "copy does not reach the released one, so the obligation is to " + "disclose, not to act.", + released=True), + + rule("R2", contribution.ALREADY_GONE, + "Nothing survives to remediate, and nothing left our hands.", storage=contribution.DESTROYED), - rule("R2", contribution.NOTIFY_ONLY, - "Immutable history — published, or already past a trust " - "boundary. Terminates remediation, not notification: you cannot " - "unpublish, so the answer is disclosure.", - terminal=True), - rule("R3", contribution.DESTROY, "Exists only because of this subject and the bytes can be " "changed. Nothing else needs it, so it goes entirely.", - exclusive=True, storage=contribution.WRITABLE), + scope=EXCLUSIVE, storage=contribution.WRITABLE), rule("R4", contribution.QUARANTINE, "Exists only because of this subject, but the storage cannot be " "written. Removal is correct and unavailable, so block use.", - exclusive=True), - - rule("R5", contribution.PURGE, - "The contribution can be isolated and the bytes can be changed. " - "Subtract it in place; the artifact survives for everyone else.", - contribution=contribution.SEPARABLE, - storage=contribution.WRITABLE), - - rule("R6", contribution.REGENERATE, - "Separable in principle but the artifact is unwritable. Rewriting " - "in place is not required — produce a fresh one without it.", - contribution=contribution.SEPARABLE), - - rule("R7", contribution.REGENERATE, - "Cannot be isolated, but the derivation can be re-executed from " - "the remaining sources.", - contribution=contribution.REGENERABLE), + scope=EXCLUSIVE), - rule("R8", contribution.QUARANTINE, - "IRREDUCIBLE: neither separable nor re-executable. Nothing can be " - "removed and nothing can be rebuilt, so block further use. Also " - "where every unrecognised class lands, by normalisation.", - contribution=contribution.IRREDUCIBLE), - ], -} + rule("R5", contribution.REGENERATE, + "The subject changed rather than left. Its part can be isolated, " + "so recompute that part and put it back; the rest stands.", + contribution=contribution.SEPARABLE, mode=TRACE), -# --------------------------------------------------------------------- v2 -# -# WHY v2 EXISTS: in v1, "does it still exist?" is asked before "was it -# published?". A published artifact whose working copy had been deleted came -# back ALREADY_GONE — "nothing to do" — which is wrong. Deleting your copy of -# something does not un-publish it. The disclosure obligation survives the -# bytes, and the same holds for material that has left under an agreement: -# our copy being gone does not reach the partner's. -# -# The practical consequence is sharper than it first looks. Because R1 is the -# only rule that can yield ALREADY_GONE, putting it first made EVERY verdict -# depend on the storage state — so under v1 nothing at all is decidable -# without a disk check. Under v2 a published artifact resolves to NOTIFY_ONLY -# whatever the disk says, because the answer genuinely does not depend on it. -# -# V1 IS LEFT EXACTLY AS IT WAS, byte for byte. Plans computed in January cite -# it and must stay replayable; editing it in place would make the version -# label a lie and the hash meaningless. A semantic change is a new version, -# never an edit. There is a test pinning v1's hash to a literal so that this -# cannot happen by accident. -# -# RULE IDS ARE STABLE ACROSS VERSIONS. R2 is the same rule here as in v1, in -# a different position — ids identify rules, not positions, so two plans on -# different versions remain comparable line by line. - -V2 = { - "version": "v2", - "description": ("Clew's remediation table. Publication is asked before " - "existence: a deleted working copy does not discharge a " - "disclosure obligation."), - "rules": [ - rule("R2", contribution.NOTIFY_ONLY, - "Immutable history — published, or already past a trust " - "boundary. Asked first, before existence: destroying our copy " - "does not reach the published or transferred one, so the " - "obligation to disclose survives the bytes.", - terminal=True), - - rule("R1", contribution.ALREADY_GONE, - "Nothing survives to remediate, and nothing left our hands. " - "Every later question presumes an artifact still exists, so this " - "is asked early — but after publication, which outlives it.", - storage=contribution.DESTROYED), - - rule("R3", contribution.DESTROY, - "Exists only because of this subject and the bytes can be " - "changed. Nothing else needs it, so it goes entirely.", - exclusive=True, storage=contribution.WRITABLE), - - rule("R4", contribution.QUARANTINE, - "Exists only because of this subject, but the storage cannot be " - "written. Removal is correct and unavailable, so block use.", - exclusive=True), - - rule("R5", contribution.PURGE, + rule("R6", contribution.PURGE, "The contribution can be isolated and the bytes can be changed. " "Subtract it in place; the artifact survives for everyone else.", - contribution=contribution.SEPARABLE, - storage=contribution.WRITABLE), + contribution=contribution.SEPARABLE, storage=contribution.WRITABLE), - rule("R6", contribution.REGENERATE, - "Separable in principle but the artifact is unwritable. Rewriting " - "in place is not required — produce a fresh one without it.", + rule("R7", contribution.REGENERATE, + "Separable in principle but the artifact is unwritable. Produce " + "a fresh one without it.", contribution=contribution.SEPARABLE), - rule("R7", contribution.REGENERATE, + rule("R8", contribution.REGENERATE, "Cannot be isolated, but the derivation can be re-executed from " "the remaining sources.", contribution=contribution.REGENERABLE), - rule("R8", contribution.QUARANTINE, - "IRREDUCIBLE: neither separable nor re-executable. Nothing can be " - "removed and nothing can be rebuilt, so block further use. Also " - "where every unrecognised class lands, by normalisation.", + rule("R9", contribution.QUARANTINE, + "Neither separable nor re-executable. Nothing can be removed and " + "nothing rebuilt, so block further use. Every unrecognised class " + "lands here, by normalisation.", contribution=contribution.IRREDUCIBLE), ], } -DEFAULT = V2 +DEFAULT = V1 -# Every policy ever shipped, so a plan citing an old version can be replayed -# under the table that was actually in force when it was computed. Entries -# here are immutable: a version is a historical record, not a place to fix -# things. -REGISTRY = {policy["version"]: policy for policy in (V1, V2)} +# Every shipped table, so a plan citing a version replays under the table +# that decided it. Entries are immutable: change the meaning, add a version. +REGISTRY = {policy["version"]: policy for policy in (V1,)} # ------------------------------------------------------------------ hashing @@ -274,12 +142,9 @@ def canonical(policy): def fingerprint(policy): """ - SHA-256 of the whole policy, description and rationales included. - - Hashing the prose as well as the logic is intentional. Two policies that - decide identically but justify differently are not the same policy: the - rationale is what an assessor reads, and a quiet edit to it changes what - the organisation is on record as having meant. + SHA-256 of the whole policy, rationales included. Two policies that + decide alike but justify differently are not the same policy: the + rationale is what an assessor reads. """ return hashlib.sha256(canonical(policy).encode("utf-8")).hexdigest() @@ -299,13 +164,10 @@ class InvalidPolicy(ValueError): def validate(policy): """ - Reject anything that could decide by accident. Returns the policy. - - Every failure here is a refusal to load, never a warning. A policy with a - typo'd dimension name would otherwise load cleanly and silently never - match, and a rule that never matches is indistinguishable from a rule that - was deleted — except that the file still shows it, so everyone believes it - is in force. + Reject anything that could decide by accident; returns the policy. Every + failure refuses to load. A rule with a mistyped dimension would + otherwise never match, and a rule that never matches looks in force + while being absent. """ if not isinstance(policy, dict): raise InvalidPolicy("policy must be an object") @@ -343,7 +205,7 @@ def validate(policy): f"rule {rule_id!r} has unknown action {action!r}; " f"known actions are {', '.join(sorted(ACTIONS))}") - if not isinstance(item.get("because"), str) or not item["because"].strip(): + if not isinstance(item.get("reason"), str) or not item["reason"].strip(): raise InvalidPolicy( f"rule {rule_id!r} needs a rationale; a rule nobody can " "explain cannot be defended when it is questioned") @@ -370,34 +232,59 @@ def load(path): return validate(json.loads(Path(path).read_text())) -def resolve_or_load(name_or_path): - """ - A shipped version name, or a path to a policy file. +POLICY_GROUP = "clew.policies" + - Version names win when both could apply: `v1` should mean the v1 everyone - else means, not a file that happens to sit in the working directory under - that name. +def available(): """ + {version: policy}: the shipped tables, then every table a provider + registered under the entry-point group. An entry loads to a table dict + or the path of a JSON file, is validated, and overrides a shipped + version of the same name. + """ + from pathlib import Path as _Path + from clew.contracts.registry import entry_points + + tables = dict(REGISTRY) + for name, dist, entry in entry_points(POLICY_GROUP): + try: + loaded = entry.load() + table = loaded if isinstance(loaded, dict) else json.loads(_Path(loaded).read_text()) + table = validate(table) + except (InvalidPolicy, OSError, ValueError, TypeError) as bad: + raise InvalidPolicy(f"policy {name!r} from {dist}: {bad}") + if table["version"] != name: + raise InvalidPolicy(f"policy {name!r} from {dist} calls itself " + f"{table['version']!r}; the entry point name and the " + "table's version must agree") + tables[name] = table + return tables + + +def resolve_or_load(name_or_path): + """A version name, shipped or registered, or a path. Names win over files.""" from pathlib import Path as _Path - if name_or_path in REGISTRY: - return resolve(name_or_path) + tables = available() + if name_or_path in tables: + return tables[name_or_path] if not _Path(name_or_path).exists(): raise InvalidPolicy( - f"{name_or_path!r} is neither a shipped version " - f"({', '.join(sorted(REGISTRY))}) nor a readable file") + f"{name_or_path!r} is neither a known version " + f"({', '.join(sorted(tables))}) nor a readable file") return load(name_or_path) def resolve(version): - """The shipped policy for a version string, for replaying an old plan.""" - if version not in REGISTRY: + """The policy for a version string, for replaying an old plan.""" + tables = available() + if version not in tables: raise InvalidPolicy( - f"unknown policy version {version!r}; shipped versions are " - f"{', '.join(sorted(REGISTRY))}. A plan citing a version this " - "build does not have cannot be replayed here — say so rather " + f"unknown policy version {version!r}; known versions are " + f"{', '.join(sorted(tables))}. A plan citing a version this " + "build does not have cannot be replayed here, say so rather " "than recomputing it under a different table.") - return REGISTRY[version] + return tables[version] # ----------------------------------------------------------------- deciding @@ -408,7 +295,7 @@ def matches(when, facts): def _english(items): - """'a, b or c' — a list a person reads, not a join artefact.""" + """'a, b or c', a list a person reads, not a join artefact.""" if len(items) == 1: return items[0] return ", ".join(items[:-1]) + " or " + items[-1] @@ -419,38 +306,24 @@ def _decide_known(facts, policy): for item in policy["rules"]: if matches(item["when"], facts): return {"action": item["action"], "rule": item["id"], - "because": item["because"]} + "reason": item["reason"]} # Guard 2, outside the rules: falling off the end is not an error and not # a pass. An incomplete policy is cautious, never permissive. return {"action": FALLTHROUGH_ACTION, "rule": FALLTHROUGH_RULE, - "because": "no rule matched; failing closed rather than deciding " - "by omission"} + "reason": "no rule matched; failing closed rather than deciding " + "by omission"} -def decide(contribution_class, storage=contribution.WRITABLE, exclusive=False, - terminal=False, policy=None): +def decide(contribution_class, storage=contribution.WRITABLE, scope=SHARED, + released=False, policy=None, mode=None): """ - Resolve one affected artifact to exactly one action, and name the rule. - - Returns {action, rule, because}. The policy's version and hash are not - repeated per decision — they belong once in the header of whatever - collects these, and hashing the policy 81 times to say the same thing - would be waste dressed up as rigour. - - `None` on storage, exclusive or terminal means NOT VERIFIED, which is - different from any real value. See the module docstring: the policy is - evaluated against every combination of the possible values, and if - they disagree no action is returned. An undetermined result has action - None and a `possible` map of the candidate actions to the rules that - would produce them. - - A value that is neither None nor one of the dimension's possible values - is an error, not a wildcard. Matching is by equality, so "writable" - would silently match no rule and fall through to QUARANTINE with a - plausible-looking citation. The contribution class is the exception, - by design: an unrecognised class is normalised to IRREDUCIBLE before - anything looks at it. + One action for one artifact, with the rule that chose it: {action, rule, + reason}. None on storage, scope, released or mode means unverified: the + policy runs once per possible value, and if they disagree action is None + with a `possible` map of the candidates. A value outside a dimension's + set is an error, not a wildcard; only the contribution class normalises, + to IRREDUCIBLE. """ policy = policy or DEFAULT @@ -459,10 +332,11 @@ def decide(contribution_class, storage=contribution.WRITABLE, exclusive=False, facts = { "contribution": contribution.normalise(contribution_class), "storage": storage, - "exclusive": exclusive, - "terminal": terminal, + "scope": scope, + "released": released, + "mode": mode, } - for field in ("storage", "exclusive", "terminal"): + for field in VERIFIABLE: value = facts[field] # The type check is not pedantry: bool is a subclass of int, so 1 # would otherwise pass as True. @@ -473,8 +347,7 @@ def decide(contribution_class, storage=contribution.WRITABLE, exclusive=False, f"({sorted(VALID[field], key=str)}); pass None if it was " "not verified") - unverified = [f for f in ("storage", "exclusive", "terminal") - if facts[f] is None] + unverified = [f for f in VERIFIABLE if facts[f] is None] if not unverified: return _decide_known(facts, policy) @@ -494,7 +367,7 @@ def decide(contribution_class, storage=contribution.WRITABLE, exclusive=False, # real answer, not a guess: it holds whatever the facts are. action, rule = next(iter(candidates.items())) return {"action": action, "rule": rule, - "because": first["because"] + "reason": first["reason"] + f" ({label} unverified, but every possible state gives " "this same answer)"} @@ -502,7 +375,7 @@ def decide(contribution_class, storage=contribution.WRITABLE, exclusive=False, "action": None, "rule": None, "possible": dict(sorted(candidates.items())), - "because": f"{label} not verified, and the verdict depends on it. " + "reason": f"{label} not verified, and the verdict depends on it. " "Verifying would decide between " + _english(sorted(candidates)) + ". Refusing to guess: assuming the artifact survives " @@ -511,13 +384,13 @@ def decide(contribution_class, storage=contribution.WRITABLE, exclusive=False, } -def remediate(contribution_class, storage=contribution.WRITABLE, - exclusive=False, terminal=False, policy=None): +def remediate(contribution_class, storage=contribution.WRITABLE, scope=SHARED, + released=False, policy=None, mode=None): """The action alone, for callers that do not need the citation. None when the verdict is undetermined. Callers that treat a falsy action as "nothing to do" are the exact failure this guards against, so anything acting on this must handle None explicitly. """ - return decide(contribution_class, storage=storage, exclusive=exclusive, - terminal=terminal, policy=policy)["action"] + return decide(contribution_class, storage=storage, scope=scope, + released=released, policy=policy, mode=mode)["action"] diff --git a/clew/ledger/query.py b/clew/ledger/query.py index 1c6c4ee..8f19903 100644 --- a/clew/ledger/query.py +++ b/clew/ledger/query.py @@ -1,47 +1,14 @@ """ -Clew core — the query surface an auditor's questions land on. - -THIS FILE KNOWS NOTHING ABOUT BIOLOGY. Subjects, triggers and fact types are -opaque strings throughout. - -WHY THIS EXISTS SEPARATELY FROM EVERYTHING ELSE ------------------------------------------------ -Two surfaces need to answer the same questions: a dashboard someone reads, -and a set of tools a language model calls on an auditor's behalf. If each -grew its own way of assembling answers they would drift, and the day they -disagreed nobody could say which one was wrong. - -So both go through here, and here has one rule. - -EVERY ANSWER CARRIES ITS CITATIONS ----------------------------------- -Not as a convention — structurally. `answer()` requires them, and there is no -path through this module that produces a conclusion without the log sequence -numbers, entry hashes, rule ids and bundle hashes a reader can go and check -independently. - -That matters most for the model-driven surface. A language model given loose -facts will produce fluent, confident, occasionally wrong prose, and an -auditor cannot tell the difference by reading it. A model given facts that -arrive welded to their citations produces prose an auditor can check line by -line, and a wrong paraphrase becomes visible rather than persuasive. - -CLEW ANSWERS WHAT IT RECORDED. IT DOES NOT ADVISE. --------------------------------------------------- -There is deliberately no query here that resolves to "you are compliant", -"this is acceptable", or "no further action is required". Every answer is a -statement about what is in the log and what the deterministic core computed -from it. Whether that satisfies an obligation is a judgement belonging to -whoever has the authority to defend it, and a tool that offered to make it -would be an attester — which this project has said, from the beginning, it -is not. - -COVERAGE TRAVELS WITH THE ANSWER --------------------------------- -Every result carries what it does NOT cover. An auditor reading a list of -three facts about a subject has no way to know whether that is the whole -history or the part that happened to be instrumented, and silence reads as -completeness. So it is said, every time, in the answer itself. +The read surface an auditor's questions land on. Subjects, triggers and fact +types are opaque strings. + +The dashboard and the MCP server both answer through here, so they cannot +drift apart. Every answer carries its citations: log sequence numbers, entry +hashes, rule ids and bundle hashes. answer() requires them. + +No query resolves to "compliant" or "no further action". Each is a statement +about what the log holds and what the core computed from it. Every result +also says what it does not cover, because silence reads as completeness. """ import json @@ -59,13 +26,8 @@ def body_of(entry): """ - An entry's body as a structure, whether it arrived parsed or as text. - - The two sources genuinely differ and both are correct. A live log hands - back parsed bodies because that is what code wants; a sealed bundle - carries the canonical TEXT, because text is what was hashed and a bundle - that re-serialised it might not re-hash to the same value. Queries should - not have to know which one they are reading. + An entry's body as a structure, whether it arrived parsed (a live log) + or as canonical text (a bundle, where the text is what was hashed). """ body = entry.get("body") if isinstance(body, str): @@ -117,7 +79,7 @@ def cite_rule(policy_document, rule_id): return {"kind": "policy_rule", "rule": rule_id, "policy_version": policy_document["version"], "policy_hash": policy_module.fingerprint(policy_document), - "action": rule["action"], "because": rule["because"]} + "action": rule["action"], "reason": rule["reason"]} return {"kind": "policy_rule", "rule": rule_id, "policy_version": policy_document["version"], "policy_hash": policy_module.fingerprint(policy_document), @@ -128,12 +90,9 @@ def cite_rule(policy_document, rule_id): def subject_history(entries, subject): """ - Everything the log holds about one subject, in the order it took effect. - - Ordered by effective_from rather than by entry order, because the - question behind this is almost always "what happened, and when", not - "what did you type, and when". Both timestamps travel with every entry so - the gap between them stays visible. + Everything the log holds about one subject, ordered by effective_from: + the question is what happened and when, not what was typed and when. + Both timestamps travel with each entry. """ matching = sorted((e for e in entries if e["subject"] == subject), key=lambda e: (instant(e["effective_from"]), e["seq"])) @@ -205,12 +164,8 @@ def policy_history(entries): def policy_in_force(entries, as_of): """ - The table in force on a date, as the log records it — not as code says. - - A plan's own header names the policy it used, and that is authoritative - for that plan. This answers the different question an assessor asks: what - was this organisation operating under at the time, according to its own - records. + The table in force on a date as the log records it, which is what an + assessor asks. A plan's own header stays authoritative for that plan. """ adoptions = [e for e in entries if e["event_type"] == POLICY_ADOPTED @@ -259,7 +214,7 @@ def verdict(plan, policy_document, task): "citations": [], "coverage": [ f"{task!r} is not in this plan. That means it was not in the " - "blast radius of this trigger — not that it is unaffected by " + "blast radius of this trigger, not that it is unaffected by " "anything.", ], } @@ -292,13 +247,9 @@ def verdict(plan, policy_document, task): "action": item.get("action"), "possible": item.get("possible"), "rule": item.get("rule"), - "because": item.get("because"), - "facts": { - "contribution": item.get("contribution"), - "storage": item.get("storage"), - "exclusive": item.get("exclusive"), - "terminal": item.get("terminal"), - }, + "reason": item.get("reason"), + "facts": {k: item.get(k) for k in + ("contribution", "storage", "scope", "released", "mode")}, "evidence_path": item.get("evidence_path"), "published_copies": item.get("published_copies"), "explanation": contribution_module.explain(item["action"]) @@ -338,13 +289,9 @@ def plan_summary(plan): def unaffected(plan, task): """ - 'Prove this task was NOT touched.' The negative question, answered. - - Worth its own query because it is the one an assessor actually asks at - submission, and because answering it well means being precise about what - was searched. A task absent from a plan is outside the blast radius of - THAT trigger, computed over THAT graph. It is not a statement about - everything that ever happened to it. + The negative question: was this task touched. A task absent from a plan + is outside that trigger's radius over that graph, and nothing more; the + answer says what was searched. """ listed = any(i["task"] == task for i in plan.get("plan", [])) return answer( diff --git a/clew/ledger/rulebook.py b/clew/ledger/rulebook.py index 6f76289..7c83c2f 100644 --- a/clew/ledger/rulebook.py +++ b/clew/ledger/rulebook.py @@ -1,23 +1,16 @@ """ -Clew — the remediation policy, from the command line. +The remediation policy from the command line. clew rulebook show clew rulebook export --out policy_v1.json clew rulebook check policy_v1.json clew rulebook register --dsn "$CLEW_DSN" --actor qa.lead@example.org -WHY REGISTER A POLICY IN THE EVENT LOG --------------------------------------- -A plan cites a policy version and hash. That is only worth something if the -claim "v1 hashed to dbb59de6... and we adopted it on this date" is itself a -recorded fact rather than something recomputed later from whatever the code -says today. So adoption is an event, with an actor and an effective date, in -the same append-only log as everything else. - -The full policy goes into the event body, not a pointer to it. A pointer to -code is worthless six months and four releases later; the log has to hold the -actual table so an old plan can be replayed even if this build no longer -ships that version. +Adoption is an event in the log with an actor and an effective date, so "v1 +hashed to dbb59de6 and we adopted it on this date" is a recorded fact rather +than something recomputed from today's code. The whole table goes into the +event body, not a pointer, so an old plan replays after the build that +shipped that version is gone. """ import argparse @@ -53,7 +46,7 @@ def cmd_show(args): for item in active["rules"]: when = ", ".join(f"{k}={v}" for k, v in sorted(item["when"].items())) print(f" {item['id']:<4} {when or '(any)':<52} -> {item['action']}") - print(f" {item['because']}") + print(f" {item['reason']}") print() trailing = 52 - (len(policy.FALLTHROUGH_RULE) - 4) print(f" {policy.FALLTHROUGH_RULE} " @@ -61,7 +54,7 @@ def cmd_show(args): print(" Outside the rule list, where no policy can remove it.") print() print("The class is normalised before matching: anything unrecognised is") - print("IRREDUCIBLE first. That fixes the facts, not the verdict — a policy") + print("IRREDUCIBLE first. That fixes the facts, not the verdict, a policy") print("that decides badly will be honoured, and will be identifiable by") print("version, hash and rule id when someone asks why.") @@ -93,7 +86,7 @@ def cmd_check(args): print(f" sha256 {stamp['policy_hash']}") print(f" {len(loaded['rules'])} rules") print() - print("Valid means well-formed and decidable — every rule can fire, names") + print("Valid means well-formed and decidable, every rule can fire, names") print("a known action, and carries a rationale. It does not mean correct.") print("Whether the table says the right thing is the author's to defend.") return 0 @@ -101,12 +94,8 @@ def cmd_check(args): def cmd_diff(args): """ - What changed between two policies, by rule id. - - Ids identify rules rather than positions, so a rule that moved is - reported as moved rather than as one deletion and one addition. Order is - semantics here — first match wins — so a pure reorder is a real change - and has to read like one. + What changed between two policies, by rule id, so a moved rule reads as + moved. First match wins, so a pure reorder is a real change. """ before, after = policy.resolve_or_load(args.before), policy.resolve_or_load(args.after) a, b = policy.identify(before), policy.identify(after) @@ -142,7 +131,7 @@ def cmd_diff(args): changes.append(f"action {old_rule['action']} -> {new_rule['action']}") if old_rule["when"] != new_rule["when"]: changes.append(f"when {old_rule['when']} -> {new_rule['when']}") - rationale = old_rule["because"] != new_rule["because"] + rationale = old_rule["reason"] != new_rule["reason"] if not changes and not rationale: continue heading = f" {rule_id}" @@ -152,8 +141,8 @@ def cmd_diff(args): if rationale: # The prose is hashed too: it is what an assessor reads, so a # changed rationale is a changed policy even when the logic holds. - print(f" - {old_rule['because']}") - print(f" + {new_rule['because']}") + print(f" - {old_rule['reason']}") + print(f" + {new_rule['reason']}") return 0 diff --git a/clew/providers.py b/clew/providers.py new file mode 100644 index 0000000..8e23f06 --- /dev/null +++ b/clew/providers.py @@ -0,0 +1,47 @@ +""" + clew providers + +Every adapter and extractor Clew can see, and the package each came from. +The check to run when your own package does not show up. +""" + +from clew.contracts import Adapter, Extractor +from clew.contracts.registry import entry_points +from clew.contracts.trigger import ENGINE_KINDS + + +def main(argv=None): + for contract in (Adapter, Extractor): + print(f"{contract.group}") + rows = sorted(entry_points(contract.group)) + if not rows: + print(" none installed") + width = max((len(name) for name, _, _ in rows), default=0) + for name, dist, entry in rows: + try: + entry.load() + state = "" if name in contract.registered else " module imported but registered nothing" + except Exception as exc: # a broken provider is what this command is for + state = f" FAILED to import: {exc}" + print(f" {name.ljust(width)} {dist}{state}") + if contract is Adapter and name in contract.registered: + for kind, trig in contract.registered[name].triggers.items(): + shadow = " shadows the engine's" if kind in ENGINE_KINDS else "" + print(f" {' ' * width} {kind}: {trig.mode.value}{shadow}") + print() + from clew.ledger import policy + print(policy.POLICY_GROUP) + registered = {n for n, _, _ in entry_points(policy.POLICY_GROUP)} + for name, dist, _ in sorted(entry_points(policy.POLICY_GROUP)): + try: + stamp = policy.identify(policy.available()[name]) + state = f"sha256 {stamp['policy_hash'][:16]}" + except policy.InvalidPolicy as bad: + state = f"REFUSED: {bad}" + print(f" {name} {dist} {state}") + for name in sorted(policy.REGISTRY): + if name not in registered: + print(f" {name} clew-lineage shipped") + print() + print("engine kinds, every graph: " + ", ".join(sorted(ENGINE_KINDS)) + ", and any label key") + return 0 diff --git a/clew/questions/drift.py b/clew/questions/drift.py index 10d9133..44fa53d 100644 --- a/clew/questions/drift.py +++ b/clew/questions/drift.py @@ -8,7 +8,7 @@ from clew.graph import blast_radius as core from clew.graph.graph import EXTERNAL -from clew.domains.nfcore import BOOKKEEPING +from clew.graph.results import BOOKKEEPING from clew.views import drift_report REPRODUCED = "REPRODUCED" @@ -71,12 +71,10 @@ def live_tasks(graph): def pair_tasks(before, after): """ - [(before hash | None, after hash | None, note)] in name order. - - One task per name on each side pairs by name. When a name repeats, - as a per-interval step does, the two sides pair on input digests, and - what that cannot settle is left unpaired with a note rather than - matched by hash order, which pairs shards crosswise. + [(before hash | None, after hash | None, note)] in name order. Unique + names pair by name. Repeated names, as per-interval steps have, pair on + input digests, and what that cannot settle stays unpaired with a note + rather than crosswise by hash order. """ names_before, names_after = live_tasks(before), live_tasks(after) pairs = [] diff --git a/clew/questions/gate.py b/clew/questions/gate.py index a1f7093..40fa854 100644 --- a/clew/questions/gate.py +++ b/clew/questions/gate.py @@ -1,39 +1,18 @@ """ -Clew — the pre-flight gate. Compliance as a build check, not a PDF. +The pre-flight gate: compliance as a build check. - clew gate --pipeline sarek --samplesheet samplesheet.csv \ - --dsn "$CLEW_DSN" --block-on Withdrawn --clear-on Reinstated \ - --out bundle/ + clew gate --pipeline sarek --samplesheet sheet.csv --dsn "$CLEW_DSN" \ + --block-on Withdrawn --clear-on Reinstated --out bundle/ -Exit 0 to proceed, 1 to stop. Run it before the pipeline, in CI, so that -using material nobody is allowed to use fails the build the way a failing -test does — at the point where it is cheap, rather than in a remediation -exercise two years later. +Exit 0 to proceed, 1 to stop. Every way of failing to establish that the +inputs are permitted exits non-zero: an unreachable log, no blocking types +given, a subject the log has never heard of (unless --allow-unknown), a +blocked subject. A green build must mean checked and permitted. -FAIL CLOSED, EVERYWHERE, ON EVERYTHING --------------------------------------- -Every way this command can fail to establish that the inputs are permitted -exits non-zero: - - the log is unreachable -> stop - no blocking types were given -> stop (a gate with nothing to block on - is not a lenient gate, it is no gate) - a subject is UNKNOWN to the log -> stop, unless --allow-unknown says - someone decided otherwise - a subject is BLOCKED -> stop - -A green build must mean "checked and permitted". If it can also mean "could -not check", the check is decorative, and a decorative compliance gate is -worse than none: it manufactures a record of diligence that did not happen. - -THE IDENTIFIER TRAP -------------------- -The samplesheet names subjects in one vocabulary and the log in another, and -nothing makes them agree. A typo, a prefix, a different column, and every -subject comes back UNKNOWN — which is why UNKNOWN stops the build by default -and why the report always states how many subjects the log had ever heard of. -A gate that goes green having recognised none of its inputs is the exact -failure this design is arranged to make loud. +The samplesheet and the log name subjects in different vocabularies and +nothing makes them agree, so the report always states how many subjects the +log recognised. A gate that goes green having recognised none of its inputs +is the failure this command is arranged to make loud. """ import argparse @@ -45,21 +24,16 @@ from clew.ledger import bundle from clew.ledger import gate as core_gate -from clew.domains import rnaseq, sarek, viralrecon - -DOMAINS = {"sarek": sarek, "viralrecon": viralrecon, "rnaseq": rnaseq} +from clew.contracts import REMOVE, Adapter, discover CHECKED = "GateChecked" def load_gate_policy(args): """ - Which fact types stop a build, and which release it. - - Deliberately not defaulted. Clew ships no opinion about what your event - types mean or which of them should stop work — that is the customer's - policy, and guessing it here would be shipping truth rather than a - template. See gate-policy.example.json. + Which fact types stop a build and which release it. No default: what + your event types mean is your policy, and a default here would be + shipping truth. See gate-policy.example.json. """ if args.gate_policy: try: @@ -98,10 +72,21 @@ def load_gate_policy(args): def main(argv=None): + argv = list(sys.argv[1:] if argv is None else argv) + adapters = discover(Adapter) + first = argparse.ArgumentParser(add_help=False) + first.add_argument("--pipeline", choices=sorted(adapters), default="sarek") + adapter = adapters[first.parse_known_args(argv)[0].pipeline] + removable = [k for k, t in adapter.triggers.items() if t.mode is REMOVE] parser = argparse.ArgumentParser( - description="Stop a pipeline run whose inputs are not permitted.") - parser.add_argument("--samplesheet", required=True) - parser.add_argument("--pipeline", choices=sorted(DOMAINS), default="sarek") + description="Stop a pipeline run whose inputs are not permitted.", + conflict_handler="resolve") + parser.add_argument("--pipeline", choices=sorted(adapters), default="sarek") + parser.add_argument("--kind", choices=sorted(adapter.triggers), + default=removable[0] if removable else None, + help="which of this pipeline's trigger kinds lists the inputs to check") + for kind in adapter.triggers.values(): + kind.add_arguments(parser) parser.add_argument("--dsn", default=os.environ.get("CLEW_DSN"), help="the event log holding the facts; $CLEW_DSN") parser.add_argument("--gate-policy", metavar="PATH", @@ -117,7 +102,7 @@ def main(argv=None): parser.add_argument("--allow-unknown", action="store_true", help="do not stop on subjects the log has never heard " "of. A real choice for a log covering part of an " - "estate — but it must be a choice.") + "estate, but it must be a choice.") parser.add_argument("--out", metavar="DIR", help="seal the result into an evidence bundle") parser.add_argument("--force", action="store_true", @@ -129,12 +114,15 @@ def main(argv=None): args = parser.parse_args(argv) policy = load_gate_policy(args) - domain = DOMAINS[args.pipeline] + if not args.kind: + raise SystemExit( + f"pipeline {args.pipeline!r} declares no trigger kind that owns inputs, " + "so there is nothing to list and check. Stopping.") try: - subjects = sorted(domain.load_subjects(args.samplesheet)) + subjects = sorted(adapter.triggers[args.kind].values(args)) except OSError as exc: raise SystemExit( - f"cannot read --samplesheet {args.samplesheet}: {exc.strerror}. " + f"cannot read the {args.kind} inputs: {exc.strerror} ({exc.filename}). " "Stopping: no inputs were checked.") if args.as_of: @@ -170,7 +158,7 @@ def main(argv=None): as_of=args.as_of, unknown_blocks=not args.allow_unknown) result["as_of_given"] = args.as_of is not None result["gate_policy"] = policy - result["samplesheet"] = str(args.samplesheet) + result["inputs"] = {"kind": args.kind, "samplesheet": getattr(args, "samplesheet", None)} result["log_head"] = log_head report(result, policy, log_head) diff --git a/clew/questions/impact.py b/clew/questions/impact.py index 4dc225e..e08f722 100644 --- a/clew/questions/impact.py +++ b/clew/questions/impact.py @@ -1,42 +1,19 @@ """ -Clew — what must happen downstream when something upstream turns out invalid. +What must happen downstream when something upstream turns out invalid. -Three triggers, one engine: + clew impact --graph g.json --trigger patient:donor_003 --samplesheet sheet.csv + clew impact --graph g.json --container gatk4 + clew impact --graph g.json --input genome.fasta + clew impact --graph g.json --pipeline qbc # whatever the adapter has pending - # consent withdrawal (a source is removed) - clew impact --graph clew/data/graph5.json --samplesheet clew/data/donors.csv --subject donor_003 +A removal takes a source away: an artifact that exists only because of it +can be destroyed, which is what scope `exclusive` means. A defect or +reference update is traced: every artifact is still wanted, so every scope +is `shared` and the worst verdict is QUARANTINE. - # tool defect (every artifact a container touched is suspect) - clew impact --graph clew/data/graph5.json --samplesheet clew/data/donors.csv --container gatk4 - - # reference / load-bearing input update - clew impact --graph clew/data/graph5.json --samplesheet clew/data/donors.csv --input genome.fasta - - # externally-asserted facts (publication) change the verdicts - ... --subject donor_003 --assertions assertions.json - -Wires the sarek domain adapter to the core traversal. All this file does is -translate between them and print the result; it holds no logic of its own. - -TWO KINDS OF TRIGGER, ONE DELIBERATE DIFFERENCE ------------------------------------------------ -A withdrawal REMOVES A SOURCE. Ownership matters: an artifact that exists -only because of the withdrawn donor has nothing left to serve, so it can be -destroyed outright. That is what `exclusive` means. - -A tool defect or reference update CASTS DOUBT. Nothing is owned by the -trigger — every affected artifact is still wanted, it just cannot be trusted. -So `exclusive` is always False for these: the worst verdict is QUARANTINE, -never DESTROY. Collapsing that distinction would delete data people need. - -WHAT THIS DOES NOT KNOW ------------------------ -Classes are assigned from pipeline evidence alone: was the script and -container recorded, and does the artifact still exist on disk. - -Publication is an assertion carried in from outside via --assertions, with -an actor and a date. MTA transfers and physical destruction are not modelled -yet. Anything unknown fails closed to IRREDUCIBLE. +Classes come from pipeline evidence alone. Publication is an assertion +carried in through --assertions with an actor and a date. Anything unknown +fails closed to IRREDUCIBLE. """ import argparse @@ -51,28 +28,22 @@ from clew.graph import contribution from clew.ledger import policy from clew.ledger.policy import UNDETERMINED -from clew.domains import rnaseq, sarek, viralrecon +from clew.contracts import Adapter, discover from clew.views import report -from clew.graph import triggers +from clew.contracts import trigger as triggers +from clew.contracts.trigger import REMOVE, TRACE, ENGINE_KINDS, Mode from clew.graph.contribution import classify from clew.graph.graph import ( MATCH_NAME_ONLY, - container_entry_nodes, container_matches, describe, - external_input_entry_nodes, external_input_matches, load_assertions, outputs_for, resolve_workdirs, ) -from clew.domains import snakemake as snakemake_domain -from clew.domains.nfcore import index_results, published_copies +from clew.graph.results import index_results, published_copies -# Which adapter translates between this pipeline's vocabulary and core's. -# Adding a pipeline = adding a module in domains/ and one entry here. -DOMAINS = {"sarek": sarek, "viralrecon": viralrecon, "rnaseq": rnaseq, - "snakemake": snakemake_domain} def graph_notes(graph): @@ -125,19 +96,9 @@ def print_notes(heading, notes): print() -def label_keys(graph): - """Every label key any task or artifact in the graph carries.""" - keys = set() - for task in graph["tasks"].values(): - keys.update(task.get("labels") or {}) - for edge in graph["edges"]: - keys.update(edge.get("labels") or {}) - return keys - - -def print_plan(domain, graph, subject, entry_nodes, affected, exclusive_set, +def print_plan(adapter, graph, subject, entry_nodes, affected, exclusive_set, published, results_index=None, active_policy=None, - work_root=None): + work_root=None, kind_name=None, mode=TRACE): """Classify every affected task and print the remediation plan.""" published_checked = (results_index is not None and bool(graph.get("output_details"))) @@ -166,9 +127,24 @@ def print_plan(domain, graph, subject, entry_nodes, affected, exclusive_set, graph, task_hash, task_hash in exclusive_set, published=published, work_root=work_root, resolved=resolved, ) - # The domain's storage check only sees the workdir. If the scratch + # The engine classifies from evidence alone. An adapter that knows the + # step (a concatenation is separable, a trained model is not) may say + # so, and the plan records who said it. + asserted = adapter.contribution(graph, task_hash, kind_name) if adapter else None + if asserted is not None: + if asserted not in contribution.CLASSES: + raise SystemExit( + f"adapter {adapter.name!r} returned {asserted!r} as the class of " + f"{task_hash}; classes are {', '.join(contribution.CLASSES)}") + if asserted != facts["contribution"]: + facts["evidence"] = (f"class {asserted} asserted by adapter {adapter.name} " + f"(evidence alone said {facts['contribution']}); " + + facts["evidence"]) + facts["contribution"] = asserted + facts["class_asserted_by"] = adapter.name + # The adapter's storage check only sees the workdir. If the scratch # copy is gone but published copies are known to exist, the artifact - # is NOT already gone — those copies are precisely what remediation + # is NOT already gone, those copies are precisely what remediation # must reach. Scratch cleanup must never launder an obligation. # A published copy IS a verified sighting, whether the scratch copy # was checked and gone or never checked at all. Those copies are @@ -178,7 +154,7 @@ def print_plan(domain, graph, subject, entry_nodes, affected, exclusive_set, and published_copies(graph, task_hash, results_index)): was = facts["storage"] facts["storage"] = contribution.WRITABLE - facts["reason"] += ("; workdir removed but published copies exist" + facts["evidence"] += ("; workdir removed but published copies exist" if was == contribution.DESTROYED else "; published copies found on disk") elif facts["storage"] == contribution.DESTROYED and not published_checked: @@ -187,14 +163,16 @@ def print_plan(domain, graph, subject, entry_nodes, affected, exclusive_set, # ALREADY_GONE for a sample whose BAM sits in results/. Leave # the dimension unverified and let the policy withhold. facts["storage"] = None - facts["reason"] += ("; workdir removed, published tree not " + facts["evidence"] += ("; workdir removed, published tree not " "checked (no --results, or no output sizes " "in the graph)") + facts["mode"] = mode.value decision = policy.decide( facts["contribution"], storage=facts["storage"], - exclusive=facts["exclusive"], - terminal=facts["terminal"], + scope=facts["scope"], + released=facts["released"], + mode=facts["mode"], policy=active_policy, ) plan.append((task_hash, facts, decision)) @@ -207,29 +185,30 @@ def print_plan(domain, graph, subject, entry_nodes, affected, exclusive_set, by_action[decision["action"] or UNDETERMINED].append( (task_hash, facts, decision)) + width = max((len(h) for h, _, _ in plan), default=4) print("REMEDIATION PLAN") + print(f" {'task':<{width}} {'process':<26} {'contribution':<12} scope") for action in sorted(by_action): rows = by_action[action] if action == UNDETERMINED: - print(f"\n {action} ({len(rows)}) — no verdict; see below") - print(f" {rows[0][2]['because']}") + print(f"\n {action} ({len(rows)}), no verdict; see below") + print(f" {rows[0][2]['reason']}") else: - print(f"\n {action} ({len(rows)}) — {contribution.explain(action)}") + print(f"\n {action} ({len(rows)}), {contribution.explain(action)}") # One rule decided this whole group; print it once with its # rationale rather than repeating an id against every task. - print(f" rule {rows[0][2]['rule']}: {rows[0][2]['because']}") + print(f" rule {rows[0][2]['rule']}: {rows[0][2]['reason']}") for task_hash, facts, _ in rows: - scope = "exclusive" if facts["exclusive"] else "shared" - print(f" {task_hash} {describe(graph, task_hash):<26} " - f"{facts['contribution']:<12} {scope}") + print(f" {task_hash:<{width}} {describe(graph, task_hash):<26} " + f"{facts['contribution']:<12} {facts['scope']}") copies = published_copies(graph, task_hash, results_index) if copies: for c in copies: flag = " AMBIGUOUS, verify before acting" if c["ambiguous"] else "" for path in c["published"]: print(f" published: {path}{flag}") - if facts["terminal"]: - print(f" {facts['reason']}") + if facts["released"]: + print(f" {facts['evidence']}") elif task_hash not in entry_nodes: # Evidence: show one concrete chain reaching this task, so the # claim is checkable rather than merely asserted. @@ -246,19 +225,56 @@ def count_undetermined(plan): return sum(1 for _, _, decision in plan if decision["action"] is None) -def plan_to_dict(domain, graph, subject, entry_nodes, plan, results_index=None, +def plan_cost(graph, plan): + """ + Per verdict, the sum of every metric the affected tasks carry, and how + many tasks carried none under that name. The names are the provider's. + Recorded figures are a floor on a rerun, never an estimate of one. + """ + by_action = {} + for task_hash, _, decision in plan: + action = decision["action"] or UNDETERMINED + bucket = by_action.setdefault(action, {"tasks": 0, "metrics": {}, "missing": {}}) + bucket["tasks"] += 1 + metrics = graph["tasks"].get(task_hash, {}).get("metrics") or {} + for name, value in metrics.items(): + bucket["metrics"][name] = round(bucket["metrics"].get(name, 0) + value, 6) + for bucket in by_action.values(): + for name in bucket["metrics"]: + bucket["missing"][name] = 0 + for task_hash, _, decision in plan: + bucket = by_action[decision["action"] or UNDETERMINED] + metrics = graph["tasks"].get(task_hash, {}).get("metrics") or {} + for name in bucket["missing"]: + if name not in metrics: + bucket["missing"][name] += 1 + return {"by_action": by_action, + "caveat": "sums of what the engine recorded for the original tasks; a rerun " + "costs at least the recorded figure, and tasks missing one add an " + "unknown amount"} + + +def print_cost(cost): + rows = [(a, b) for a, b in cost["by_action"].items() if b["metrics"]] + if not rows: + return + print("\nCOST OF THIS PLAN") + for action, bucket in sorted(rows): + parts = [f"{value:g} {name}" for name, value in sorted(bucket["metrics"].items())] + gaps = [f"{n} without {name}" for name, n in sorted(bucket["missing"].items()) if n] + line = f" {action} {bucket['tasks']} task(s): " + ", ".join(parts) + if gaps: + line += "; " + ", ".join(gaps) + print(line) + print(f" {cost['caveat']}") + + +def plan_to_dict(adapter, graph, subject, entry_nodes, plan, results_index=None, active_policy=None, notes=()): """ - The remediation plan as data, for scripts and CI rather than eyes. - - Deliberately clock-free: the same inputs must produce byte-identical - output, because "re-run it and get the same answer" is the whole basis - of Clew's evidence claim. Whoever stores this can wrap it with a - timestamp; Clew itself only states what follows from the inputs. - - Carries the policy version AND its hash. The version alone is a label - anyone can print; the hash is what makes two parties able to prove they - were reading the same table. + The plan as data for scripts and CI. Clock-free, so the same inputs give + byte-identical output. Carries the policy version and its hash; the hash + is what lets two parties prove they read the same table. """ forward = core.forward_index(graph["edges"]) tree = core.evidence_tree(entry_nodes, forward) @@ -275,17 +291,21 @@ def plan_to_dict(domain, graph, subject, entry_nodes, plan, results_index=None, # an estimate: a fan-out landing on a cluster is expensive # whatever a clock says. Empty for engines that run one machine. "target": task.get("target", ""), + **({"metrics": task["metrics"]} if task.get("metrics") else {}), # None when undetermined. A consumer treating a falsy action as # "nothing to do" is the exact failure this guards against, so # `possible` is present precisely when `action` is not. "action": action, "rule": decision["rule"], - "because": decision["because"], + "reason": decision["reason"], "contribution": facts["contribution"], "storage": facts["storage"], - "exclusive": facts["exclusive"], - "terminal": facts["terminal"], - "reason": facts["reason"], + "scope": facts["scope"], + "released": facts["released"], + "mode": facts["mode"], + "evidence": facts["evidence"], + **({"class_asserted_by": facts["class_asserted_by"]} + if "class_asserted_by" in facts else {}), } # What a re-run script needs, for the artifacts it must rebuild. if action == "REGENERATE": @@ -308,13 +328,14 @@ def plan_to_dict(domain, graph, subject, entry_nodes, plan, results_index=None, counts[decision["action"] or UNDETERMINED] += 1 return { - "clew_plan_version": 1, + "clew_plan_version": 2, **policy.identify(active_policy), "trigger": subject, "entry_tasks": sorted(entry_nodes), "tasks_total": len(graph["tasks"]), "tasks_affected": len(plan), "actions": dict(sorted(counts.items())), + "cost": plan_cost(graph, plan), "plan": items, "caveats": [ "classes assigned from pipeline evidence only " @@ -324,7 +345,7 @@ def plan_to_dict(domain, graph, subject, entry_nodes, plan, results_index=None, "UNDETERMINED items are not clean; they are unanswered. Re-run " "with --work-root and --results where the artifacts live to " "settle them", - "publication status is an external assertion, not verified by Clew", + "release status is an external assertion, not verified by Clew", "MTA transfers and physical destruction are not modelled", "uninstrumented systems are unknown, never clean", # What the graph and the trigger said about their own limits, @@ -334,73 +355,160 @@ def plan_to_dict(domain, graph, subject, entry_nodes, plan, results_index=None, } -def main(argv=None): - parser = argparse.ArgumentParser(description="Compute a blast radius and remediation plan.") - parser.add_argument("--graph", required=True, help="graph JSON from an extractor") - parser.add_argument( - "--samplesheet", - help="nf-core samplesheet CSV. Needed only for --subject:\n" - "is the one thing a domain has to resolve.") - parser.add_argument("--pipeline", choices=sorted(DOMAINS), default="sarek", - help="which domain adapter reads the samplesheet and names") +def per_trigger_path(path, trigger): + """plan.json -> plan.subject-SPC-0412.json""" + if not path or path == "-": + return path + p = Path(path) + tag = f"{trigger['kind']}-{trigger['value']}".replace("/", "_") + return str(p.with_name(f"{p.stem}.{tag}{p.suffix}")) + + +def pending_triggers(adapter): + """The adapter's pending triggers, checked before any is asked.""" + from clew.contracts.adapter import check + pending = adapter.pending() + if not isinstance(pending, list): + raise SystemExit(f"{adapter.name}: pending() must return a list") + for i, trigger in enumerate(pending): + problems = check(trigger) + if problems: + raise SystemExit(f"{adapter.name}: pending trigger {i}: {'; '.join(problems)}") + return pending + + +def answer_each(args, adapter, argv): + """One plan per pending trigger. A bad trigger fails its answer, not the batch.""" + triggers_to_ask = pending_triggers(adapter) + failed = [] + for trigger in triggers_to_ask: + spec = f"{trigger['kind']}:{trigger['value']}" + who = trigger.get("asserted_by") + when = trigger.get("date") + print("=" * 70) + print(f"TRIGGER {spec}" + (f" asserted by {who}" if who else "") + + (f" on {when}" if when else "")) + print("=" * 70) + call = list(argv) + ["--trigger", spec] + for flag, value in (("--json", args.json_out), ("--html", args.html_out)): + if value: + call = [a for a in call if a != flag and a != value] + call += [flag, per_trigger_path(value, trigger)] + try: + main(call) + except SystemExit as stop: + failed.append((spec, str(stop))) + print(f"FAILED {spec}: {stop}", file=sys.stderr) + print() + print(f"{len(triggers_to_ask)} triggers, {len(failed)} failed") + return 1 if failed else 0 + + +def parser_for(adapters, adapter): + parser = argparse.ArgumentParser( + description="Compute a blast radius and remediation plan.", + conflict_handler="resolve") + parser.add_argument("--graph", help="graph JSON from an extractor") + parser.add_argument("--runs", metavar="DIR", + help="the engine's record instead of --graph, read through " + "whichever installed extractor recognises it; with --json, " + "the derived graph is written beside the plan") + parser.add_argument("--run", help="which run under --runs; default: the latest") + parser.add_argument("--pipeline", choices=sorted(adapters), default="sarek", + help="which adapter's trigger kinds apply") trigger = parser.add_mutually_exclusive_group() - trigger.add_argument( - "--subject", "--donor", dest="subject", - help="withdraw a subject: a donor, a batch, a specimen. What " - "one is belongs to the domain adapter, not here. --donor " - "is the former name and still works.") - trigger.add_argument("--container", help="flag every task run in a matching container") trigger.add_argument( "--trigger", - help="kind:value, for example container:gatk4, script:prep.py, " - "input:genome.fa or subject:batch_017. An unknown kind is " - "read as a label key, so a graph that carries " - "labels: {tissue: liver} answers tissue:liver with no " - "adapter and no new flag.") - trigger.add_argument("--input", dest="input_file", - help="invalidate an external input file by basename") - parser.add_argument("--mode", choices=("remove", "distrust"), - help="what kind of wrong: remove = the source must be " - "taken out (withdrawal; exclusive artifacts can be " - "destroyed); distrust = the data is suspect but " - "still wanted (contamination, defects; worst case " - "quarantine). Defaults: remove for --donor, " - "distrust for --container/--input.") + help="kind:value, for example container:gatk4, input:genome.fa or " + "patient:donor_003. Kinds are this pipeline's, then the engine's " + "(container, script, process, input), then any label key the " + "graph carries. A bare kind prints the reach of every value.") + trigger.add_argument("--container", help="short for --trigger container:X") + trigger.add_argument("--input", dest="input_file", help="short for --trigger input:X") + parser.add_argument("--mode", choices=("remove", "trace"), + help="remove = the source is removed and what only it " + "fed can be destroyed; trace = follow what it touched, " + "everything stays, worst case quarantine. Defaults " + "to what the kind declares.") parser.add_argument("--assertions", help="JSON file of externally-asserted facts") parser.add_argument("--policy", metavar="VERSION|PATH", - help="a shipped policy version (v1, v2) or a policy " - "JSON file. Defaults to the current table. Pin " - "this to replay a historical plan under the " - "table that was in force when it was computed.") + help="a shipped policy version (v1) or a policy JSON " + "file. Defaults to the current table.") parser.add_argument("--files", action="store_true", help="list affected output files") parser.add_argument("--html", dest="html_out", metavar="PATH", - help="write one self-contained HTML page, " - "or - for stdout") + help="write one self-contained HTML page, or - for stdout") parser.add_argument("--json", dest="json_out", metavar="PATH", help="also write the plan as JSON ('-' for stdout)") parser.add_argument("--work-root", metavar="DIR", - help="the run's work directory, so Clew can check " - "whether each task's artifacts still exist. " - "Without it no storage claim is made, and any " - "verdict that depends on storage is reported " - "UNDETERMINED rather than guessed.") + help="the run's work directory, so storage can be checked; " + "without it storage-dependent verdicts are UNDETERMINED") parser.add_argument("--results", metavar="DIR", - help="the run's published results directory. Needed " - "with --work-root: a task whose scratch was " - "cleaned is only ALREADY_GONE once its published " - "copies were also looked for. Plan items name " - "the copies found. Needs a graph that records " - "output sizes (store, work or horus extractors).") - args = parser.parse_args(argv) - - domain = DOMAINS[args.pipeline] + help="the run's published results directory, checked with " + "--work-root before anything reads ALREADY_GONE") + for kind in adapter.triggers.values(): + kind.add_arguments(parser) + return parser + + +def spec_of(args): + if args.trigger: + return args.trigger + if args.container: + return f"container:{args.container}" + if args.input_file: + return f"input:{args.input_file}" + return None + + +def print_reach(graph, entry): + """One line per value of the kind: how far each reaches.""" + radius = core.blast_radius(graph, entry) + print(f"{len(graph['tasks'])} tasks, {len(graph['edges'])} edges, {len(entry)} values\n") + print(f"{'value':<12} {'entry':>6} {'affected':>9} {'exclusive':>10} {'shared':>7}") + for value in sorted(radius): + r = radius[value] + print(f"{value:<12} {len(entry[value]):>6} {len(r['affected']):>9} " + f"{len(r['exclusive']):>10} {len(r['shared']):>7}") + + +def main(argv=None): + argv = list(sys.argv[1:] if argv is None else argv) + adapters = discover(Adapter) + first = argparse.ArgumentParser(add_help=False) + first.add_argument("--pipeline", choices=sorted(adapters), default="sarek") + chosen, _ = first.parse_known_args(argv) + adapter = adapters[chosen.pipeline] + args = parser_for(adapters, adapter).parse_args(argv) + + spec = spec_of(args) + if not spec: + if adapter.pending(): + return answer_each(args, adapter, argv) + raise SystemExit( + "nothing to ask: give --trigger kind:value, or --container / --input. " + f"This pipeline's kinds: {', '.join(adapter.triggers) or 'none'}; the " + f"engine's: {', '.join(ENGINE_KINDS)}. A bare --trigger kind prints " + "the reach of every value.") + try: active_policy = (policy.resolve_or_load(args.policy) if args.policy else policy.DEFAULT) except policy.InvalidPolicy as bad: # Refuse to compute rather than compute under a table nobody vetted. raise SystemExit(f"policy rejected: {bad}") - graph = core.load_graph(args.graph) + if bool(args.graph) == bool(args.runs): + raise SystemExit("give --graph or --runs, not both") + if args.runs: + from clew.extract.runs import Runs + graph = Runs(args.runs).load(args.run) + if args.json_out and args.json_out != "-": + # The plan must be re-derivable from a file someone can hash and + # hand over, not from a directory that may have changed since. + derived = Path(args.json_out).with_suffix(".graph.json") + derived.write_text(json.dumps(graph, indent=2)) + print(f"graph derived from {args.runs} ({graph['run']['name']}) written to {derived}\n") + else: + graph = core.load_graph(args.graph) results_index = index_results(args.results) if args.results else None if args.results and not graph.get("output_details"): print("note: this graph records no output sizes, so published " @@ -409,163 +517,90 @@ def main(argv=None): notes = graph_notes(graph) print_notes("WHAT THIS GRAPH DOES NOT COVER", notes) - if args.trigger: - kind, value = triggers.parse(args.trigger) - if kind == "subject" and args.samplesheet: - # A samplesheet is how a subject resolves on an nf-core graph, - # whose tasks carry no labels. Same path as --subject, so the - # documented spelling and the flag give one answer. - args.subject, args.trigger = value, None - elif kind == "container": - args.container = args.container or value - # Only a subject trigger needs a domain to resolve one. Container and - # input triggers are graph questions, so asking one should not require - # naming a pipeline or producing its samplesheet. A samplesheet with no - # subject asks for the per-subject table, so it is loaded whenever given. - if args.subject and not args.samplesheet: + kind_name, value = triggers.parse(spec) + kind = triggers.lookup(adapter, kind_name, graph) + if kind is None: raise SystemExit( - "--subject needs --samplesheet: resolving a subject to the " - "nodes it enters at is the one thing a domain does.") - doubt = args.trigger or args.container or args.input_file - if not doubt and not args.samplesheet: + f"unknown trigger kind {kind_name!r}: not one this pipeline declares " + f"({', '.join(adapter.triggers) or 'none'}), not an engine kind " + f"({', '.join(ENGINE_KINDS)}), and no task or edge in this graph " + f"carries a {kind_name!r} label.") + mode = Mode(args.mode) if args.mode else kind.mode + if mode is REMOVE and kind.mode is not REMOVE: raise SystemExit( - "nothing to ask: give --container, --input or --trigger, or " - "--samplesheet with --subject. --samplesheet alone prints the " - "reach of every subject.") - donors = domain.load_subjects(args.samplesheet) if args.samplesheet else {} - published = load_assertions(args.assertions) + f"--mode remove needs a kind that owns something; {kind_name!r} can only " + "be traced, it does not remove a source.") - # --- doubt triggers: single subject, nothing exclusive ------------------- - if args.trigger or args.container or args.input_file: - if args.mode == "remove": - # Removal needs an owner: "exclusive" only means something when - # other subjects exist to compare against. A retracted upstream - # dataset is a real remove-shaped input trigger, but computing - # its exclusive set needs multi-root traversal we don't do yet. - raise SystemExit( - "--mode remove requires a subject trigger (--subject); " - "container and input triggers cast doubt, they do not remove " - "an owned source.") - if args.trigger: - kind, value = triggers.parse(args.trigger) - subjects = triggers.resolve(graph, kind, value) - if not next(iter(subjects.values())): - if kind in triggers.KINDS or kind in label_keys(graph): - raise SystemExit(f"no task matches {args.trigger!r}.") - hint = (" For an nf-core run the subject comes from the " - "samplesheet: --subject X --samplesheet sheet.csv, " - "or this trigger with --samplesheet." - if kind == "subject" else "") - raise SystemExit( - f"no task matches {args.trigger!r}: nothing in this " - f"graph carries a {kind!r} label, so a {kind}: trigger " - f"cannot be resolved from the graph alone.{hint}") - elif args.container: - subjects = container_entry_nodes(graph, args.container) - else: - subjects = external_input_entry_nodes(graph, args.input_file) + entry = kind.resolve(graph, value, args) + if value is None: + print_reach(graph, entry) + return + published = load_assertions(args.assertions) - subject, entry_nodes = next(iter(subjects.items())) + if kind.mode is TRACE: + subject, entry_nodes = next(iter(entry.items())) if not entry_nodes: - raise SystemExit(f"no tasks match {subject}") - input_name = args.input_file or (value if args.trigger and kind == "input" else None) - trigger_notes = (container_notes(graph, args.container) if args.container - else input_notes(graph, input_name) if input_name else []) + raise SystemExit(f"no task matches {subject}") + trigger_notes = (container_notes(graph, value) if kind_name == "container" + else input_notes(graph, value) if kind_name == "input" else []) print_notes("TRIGGER NOTES", trigger_notes) notes = notes + trigger_notes + radius = core.blast_radius(graph, entry) + affected, exclusive, label = radius[subject]["affected"], set(), subject + else: + if not entry[value]: + # Zero entry nodes is a failed attribution, not a clean result. + tagged = sum(len(nodes) for nodes in entry.values()) + hint = (f" No task matched ANY {kind_name}: the ids and the run disagree." + if not tagged else "") + raise SystemExit(f"{value!r} is a known {kind_name} but no task in the " + f"graph carries it. Not attributable, not clean.{hint}") + radius = core.blast_radius(graph, entry) + result = radius[value] + entry_nodes = entry[value] + if mode is REMOVE: + # Withdrawal: what only this value fed has nothing left to serve. + label, exclusive = f"removal of {value}", result["exclusive"] + else: + # Contamination, swap, QC failure: the data is wrong, not removed. + label, exclusive = f"trace of {value}", set() + affected = result["affected"] - radius = core.blast_radius(graph, subjects) - affected = radius[subject]["affected"] - # Doubt, not removal: every artifact is still wanted. See header. - plan = print_plan(domain, graph, subject, entry_nodes, affected, - exclusive_set=set(), published=published, - results_index=results_index, - active_policy=active_policy, - work_root=args.work_root) - print_caveats(bool(published), active_policy, - undetermined=count_undetermined(plan)) - # Last on stdout on purpose: with --json -, a consumer can split at - # the final '{' and parse cleanly. - write_outputs(args.json_out, args.html_out, domain, graph, subject, entry_nodes, plan, - results_index, active_policy, notes) - return - - # --- withdrawal: exclusive/shared computed against the other donors ------ - entry = domain.subject_entry_nodes(graph, donors) - radius = core.blast_radius(graph, entry) - - if not args.subject: - print(f"{len(graph['tasks'])} tasks, {len(graph['edges'])} edges, " - f"{len(donors)} subjects\n") - print(f"{'subject':<12} {'entry':>6} {'affected':>9} {'exclusive':>10} {'shared':>7}") - for donor in sorted(radius): - r = radius[donor] - print(f"{donor:<12} {len(entry[donor]):>6} {len(r['affected']):>9} " - f"{len(r['exclusive']):>10} {len(r['shared']):>7}") - return + plan = print_plan(adapter, graph, label, entry_nodes, affected, exclusive, + published, results_index=results_index, + active_policy=active_policy, work_root=args.work_root, + kind_name=kind_name, mode=mode) - if args.subject not in radius: - raise SystemExit(f"unknown subject {args.subject!r}; known: {', '.join(sorted(radius))}") - - result = radius[args.subject] - if not entry[args.subject]: - # Zero entry nodes is a failed attribution, not a clean result. The - # samplesheet id matched no task tag, which is what an id mismatch - # between LIMS, samplesheet and pipeline looks like. - tagged = sum(len(nodes) for nodes in entry.values()) - hint = (" No task tag matched ANY subject in the samplesheet: the " - "ids in the samplesheet and the tags in the run disagree." - if not tagged else "") - raise SystemExit( - f"{args.subject!r} is in the samplesheet but no task in the " - f"graph carries its tag. Not attributable, not clean.{hint}") - mode = args.mode or "remove" - if mode == "remove": - # Withdrawal: the subject's exclusive artifacts have nothing left to - # serve and can be destroyed. - label = f"withdrawal of {args.subject}" - exclusive = result["exclusive"] - else: - # Contamination / swap / QC failure: the subject's data is WRONG, not - # withdrawn. Every artifact is still wanted once the cause is fixed, - # so nothing is owned-and-destroyable; worst case is quarantine. - label = f"distrust of {args.subject}" - exclusive = set() - plan = print_plan(domain, graph, label, entry[args.subject], - result["affected"], exclusive, published, - results_index=results_index, active_policy=active_policy, - work_root=args.work_root) - - if args.files: + if args.files and kind.mode is REMOVE: exclusive_files = outputs_for(graph, result["exclusive"]) shared_files = outputs_for(graph, result["shared"]) - print(f"\nFILES exclusive to {args.subject}: {len(exclusive_files)}") + print(f"\nFILES exclusive to {value}: {len(exclusive_files)}") for path in exclusive_files[:20]: print(f" {path}") if len(exclusive_files) > 20: print(f" ... {len(exclusive_files) - 20} more") - print(f"\nFILES shared with other donors: {len(shared_files)}") + print(f"\nFILES shared with others: {len(shared_files)}") for path in shared_files[:20]: print(f" {path}") if len(shared_files) > 20: print(f" ... {len(shared_files) - 20} more") - print_caveats(bool(published), active_policy, - undetermined=count_undetermined(plan)) - # Last on stdout on purpose: with --json -, a consumer can split at the - # final '{' and parse cleanly. - write_outputs(args.json_out, args.html_out, domain, graph, label, entry[args.subject], plan, + print_cost(plan_cost(graph, plan)) + print_caveats(bool(published), active_policy, undetermined=count_undetermined(plan), + asserted=sum(1 for _, facts, _ in plan if 'class_asserted_by' in facts)) + # Last on stdout on purpose: with --json -, a consumer can split at the final '{'. + write_outputs(args.json_out, args.html_out, adapter, graph, label, entry_nodes, plan, results_index, active_policy, notes) -def write_outputs(json_out, html_out, domain, graph, subject, entry_nodes, +def write_outputs(json_out, html_out, adapter, graph, subject, entry_nodes, plan, results_index=None, active_policy=None, notes=()): """ Render the plan in whichever formats were asked for, building it once. """ if not json_out and not html_out: return - built = plan_to_dict(domain, graph, subject, entry_nodes, plan, + built = plan_to_dict(adapter, graph, subject, entry_nodes, plan, results_index, active_policy, notes) if html_out: report.write(built, html_out) @@ -584,18 +619,22 @@ def write_json(json_out, built): print(f"\nwrote {json_out}") -def print_caveats(have_assertions, active_policy=None, undetermined=0): +def print_caveats(have_assertions, active_policy=None, undetermined=0, asserted=0): stamp = policy.identify(active_policy or policy.DEFAULT) print("\n" + "-" * 60) print(f"Computed under policy {stamp['policy_version']}, " f"sha256 {stamp['policy_hash']}.") - print("Classes assigned from pipeline evidence only (script + container " - "recorded, artifact present on disk).") + if asserted: + print(f"{asserted} class(es) asserted by the adapter, recorded per item; the rest " + "from pipeline evidence (script + container recorded, artifact on disk).") + else: + print("Classes assigned from pipeline evidence only (script + container " + "recorded, artifact present on disk).") if have_assertions: print("Publication status from the assertions file; recorded as an " "external claim with actor and date, not verified by Clew.") else: - print("No assertions file given: publication status unknown, all " + print("No assertions file given: release status unknown, all " "artifacts treated as unpublished.") print("MTA transfers and physical destruction are not modelled here.") if undetermined: diff --git a/clew/questions/reclaim.py b/clew/questions/reclaim.py index 9a28683..d671e4d 100644 --- a/clew/questions/reclaim.py +++ b/clew/questions/reclaim.py @@ -13,7 +13,7 @@ from clew.graph import blast_radius as core from clew.graph.graph import (EXTERNAL, STATUS_FAILED, STATUS_UNKNOWN, published_digests, resolve_workdirs, task_status) -from clew.domains.nfcore import BOOKKEEPING +from clew.graph.results import BOOKKEEPING from clew.views import reclaim_report from clew.views.reclaim_report import human @@ -149,14 +149,11 @@ def can_recompute(self, task_hash, seen=()): def published_copy(self, path, detail): """ How the published tree holds this output: ('digest' | 'hardlink', - [paths]) when it does, ('symlink', [paths]) when it only points into - work, ('changed', [paths]) when a copy is no longer the size that - was digested, ('none', []) when it does not, or ('unverified', - [paths]) when --results was not given to check. - - Two outputs can share a digest (an empty file, a repeated header), - so a copy with the output's own basename is preferred; only when - none has it are all copies with that digest considered. + [paths]), ('symlink', [paths]) when it only points into work, + ('changed', [paths]) when a copy no longer has the digested size, + ('none', []), or ('unverified', [paths]) without --results. A copy + with the output's own basename is preferred, since two outputs can + share a digest. """ rels = self.published.get(detail.get("digest") or "", []) named = [rel for rel in rels if Path(rel).name == Path(detail["file"]).name] diff --git a/clew/views/dashboard.py b/clew/views/dashboard.py index 5c02161..445c2ab 100644 --- a/clew/views/dashboard.py +++ b/clew/views/dashboard.py @@ -1,39 +1,17 @@ """ -Clew — a self-contained HTML view over an evidence store. +One self-contained HTML page over an evidence store. clew dashboard --bundles /path/to/bundles --out evidence.html -One file, no server, no network, no scripts. An auditor opens it from a USB -stick on a machine with no access to anything, and it still works. Printing -it produces something usable, because auditors print things. - -THIS PAGE IS NOT THE RECORD ---------------------------- -The bundles are. This is a view generated from them, and it says so at the -top, because a rendered summary is exactly the kind of artifact that gets -detached from its source and quoted years later. Every panel carries the -bundle hash it was drawn from so the page can always be traced back to -something checkable, and the integrity panel reports the verifier's own -output rather than a rendering of it. - -It shares core/query.py with the MCP server on purpose. Two surfaces -answering the same questions two different ways would eventually disagree, -and on the day they did nobody could say which was wrong. - -WHAT IS NOT KNOWN IS GIVEN THE SAME WEIGHT AS WHAT IS ------------------------------------------------------ -A compliance dashboard that renders gaps in small grey text below the fold is -worse than no dashboard: it manufactures the impression of a clean bill of -health out of an incomplete record. So coverage limits, undetermined verdicts -and subjects the log has never heard of appear near the top, in the same -visual weight as everything else, and the summary counts them explicitly. - -NO CLOCK --------- -The page carries no generation timestamp, so regenerating it from unchanged -bundles produces an identical file. Two auditors comparing pages should be -comparing evidence, not diffing dates. The bundles' own hashes and the log -heads they anchor to are the identity of what is shown. +No server, no network, no scripts, prints legibly. The page is a view and +the bundles are the record: every panel carries the hash of the bundle it +came from. It shares query.py with the MCP server so the two cannot +disagree. + +Coverage limits, undetermined verdicts and subjects the log has never heard +of appear near the top at full weight. A dashboard that hides gaps below the +fold manufactures a clean bill of health. No generation timestamp, so +unchanged bundles render to identical bytes. """ import argparse @@ -124,7 +102,7 @@ def section_header(store, root): parts = [ "

Clew evidence

", '

A view generated from the bundles below. ' - 'This page is not the record — the bundles are, and ' + 'This page is not the record, the bundles are, and ' 'each panel names the bundle hash it was drawn from so anything here ' 'can be traced back and checked independently with ' 'clew evidence verify.

', @@ -155,7 +133,7 @@ def section_integrity(store): else "unknown") label = check["check"] if check["ok"] else ( f"{check['check']} FAILED" if check["ok"] is False - else f"{check['check']} —") + else f"{check['check']} , ") cells.append(f'{tag(label, kind)} ' f'{esc(check["detail"])}') rows.append( @@ -192,7 +170,7 @@ def section_unknowns(store): if missing: notes.append( f"{bundle['name']}: {len(missing)} of " - f"{len(plan.get('plan', []))} items have no verdict — " + f"{len(plan.get('plan', []))} items have no verdict, " "storage was not verified and the answer depends on it.") gate = bundle["documents"].get("gate.json") if gate: @@ -248,7 +226,7 @@ def section_log(store): return ("

The log

" "

Two clocks. Effective is when a decision was " "made in the world; recorded is when it reached " - "the log. Where they differ, both matter — a fact effective in " + "the log. Where they differ, both matter, a fact effective in " "March and recorded in August means work done in between was done " "in good faith and still has to be accounted for.

" "" @@ -278,7 +256,7 @@ def section_policy(store): for a in adoptions) return ("

Policy

" "

Which remediation table was in force, and from when. The hash " - "is what makes the version label checkable — two parties can prove " + "is what makes the version label checkable, two parties can prove " "they were reading the same table.

" "
SeqEffectiveRecorded
" "" + rows + "
VersionEffective fromAdopted bysha256
" @@ -302,7 +280,7 @@ def section_plan(bundle, store): kind = "unknown" if not detail["action"] else ( "bad" if action in ("DESTROY", "QUARANTINE") else "") chain = " → ".join(detail.get("evidence_path") or []) - because = detail["because"] if detail["action"] else ( + because = detail["reason"] if detail["action"] else ( "no verdict: storage was not verified and the answer depends on " "it. Possible: " + ", ".join(sorted(detail.get("possible") or {}))) rows.append( @@ -340,7 +318,7 @@ def section_gate(bundle): for subject, detail in sorted(result.get("subjects", {}).items())) verdict = ("PASS" if result.get("passed") else "STOP") - return (f"

Gate — {esc(result.get('samplesheet'))} " + return (f"

Gate, {esc(result.get('samplesheet'))} " f"{tag(verdict, 'ok' if result.get('passed') else 'bad')}

" f'

bundle {esc(bundle["hash"])}
' f"as of {esc(result.get('as_of') or 'all facts in effect')}, " @@ -382,7 +360,7 @@ def render(store, root): inputs. Anyone can re-run it and get the same answer.

Clew claims nothing about whether the inputs were true or the policy was correct. Those belong to whoever has the domain authority -to defend them. This is a system of record, not an attester — it does not +to defend them. This is a system of record, not an attester, it does not decide whether a use was compliant, it makes it impossible to lose the record of what was decided, on what basis, and when.

It does not prove physical destruction. No cryptography reaches a freezer. @@ -406,7 +384,7 @@ def main(argv=None): print(f"wrote {args.out} ({len(store[0])} bundles, " f"{len(store[1])} log entries)") if store[2]: - print(f"WARNING: {len(store[2])} sequence conflicts — these bundles " + print(f"WARNING: {len(store[2])} sequence conflicts, these bundles " f"were sealed from different logs; the page says so.") diff --git a/clew/views/fonts.py b/clew/views/fonts.py index d643630..c11b0fb 100644 --- a/clew/views/fonts.py +++ b/clew/views/fonts.py @@ -1,17 +1,8 @@ """ -Clew — the QuietFlare brand faces, embedded. - -Inter and Inter Tight, latin subset, as woff2 data URIs. The site loads -these from a font host; a report cannot, because a page that fetches -anything stops opening on a machine with no network, and these get -emailed and archived. - -Three weights, not five: Inter 400 for running text, Inter 500 for -labels and tags, Inter Tight 800 for headings. Roughly 185 KB of the -page, which is the price of the page looking like QuietFlare wherever -it is opened. - -Regenerate with the script in docs/ if the site's faces change. +The QuietFlare faces, embedded: Inter 400 and 500 and Inter Tight 800, latin +subset, as woff2 data URIs, about 185 KB. A report that fetches fonts stops +opening offline, and these get emailed and archived. Regenerate with the +script in docs/ if the site's faces change. """ FACES = """ diff --git a/clew/views/mcp_server.py b/clew/views/mcp_server.py index b8bd104..20000d8 100644 --- a/clew/views/mcp_server.py +++ b/clew/views/mcp_server.py @@ -1,51 +1,18 @@ """ -Clew — an MCP server, so an auditor can ask questions in their own words. +An MCP server over sealed bundles, so an auditor can ask in their own words. clew mcp --bundles /path/to/bundles -Speaks MCP over stdin/stdout as newline-delimited JSON-RPC 2.0. No SDK, no -dependency: the protocol is small enough that adding one would cost more than -it saved. - -WHY THIS IS A SERVER AND NOT A CHATBOT --------------------------------------- -Clew ships no model and calls none. It exposes tools; the auditor's own MCP -client supplies the conversation. That is not modesty about scope, it is the -architecture: - - - Nothing here can be talked into a different answer. The tools compute - from the log and the sealed bundles, deterministically, and a model - calling them cannot change what comes back. - - The auditor's organisation chooses the model, and keeps whatever - controls it already has over which models may see its data. - - "No AI in the decision path" stays literally true. There is no path from - a model's output back into a verdict — verdicts were computed before - this process started, by code that has never seen a prompt. - -The model's job is to find the right evidence and read it out. It is a -skilled index, not a witness. - -READ-ONLY BY CONSTRUCTION -------------------------- -No tool here writes anything, and the server never opens a connection that -could. It reads sealed bundles from a directory. Recording a fact is -clew log, run by a person with an actor identity, and it stays that way — -an auditor's chat session is the last place a new fact should be able to -enter a compliance record. - -WHAT THE MODEL IS TOLD, AND WHY IT IS TOLD IT HERE --------------------------------------------------- -The `instructions` returned at initialize are the guardrail. A model handed -loose facts about compliance will produce fluent, confident, occasionally -wrong prose, and an auditor cannot tell the difference by reading it. So -every tool returns facts welded to their citations, and the instructions say -plainly: quote them, never conclude compliance, and say when you do not know. - -That is a guardrail, not a guarantee. A model can still paraphrase badly. -What the design buys is that a bad paraphrase sits next to the log sequence -number and rule id that contradict it, so it is checkable rather than merely -persuasive. Anyone deploying this should assume the prose is a convenience -and the citations are the record. +JSON-RPC 2.0 over stdin/stdout, no SDK. Clew ships no model and calls none; +the auditor's client supplies the conversation. The tools compute from the +bundles deterministically, so a model cannot change what comes back, and no +path leads from its output into a verdict. Nothing here writes: recording a +fact is clew log, run by a person with an actor identity. + +Every tool returns facts with their citations, and the initialize +instructions tell the model to quote them, never conclude compliance, and +say when it does not know. A bad paraphrase then sits next to the log +sequence number and rule id that contradict it. """ import argparse @@ -70,7 +37,7 @@ identified by a policy version, a rule id and a hash. You find those and read them out. You do not produce verdicts, and you cannot change one. -ALWAYS QUOTE THE CITATIONS. Every result carries a `citations` list — log +ALWAYS QUOTE THE CITATIONS. Every result carries a `citations` list, log sequence numbers, entry hashes, rule ids, policy hashes. Put them in your answer. An auditor must be able to leave your reply and go and check it in the record without asking you anything further. An answer without its @@ -85,7 +52,7 @@ READ `coverage` OUT LOUD. Every result carries what it does not cover. An empty list of facts about a subject means nothing was recorded under that -identifier — not that nothing happened. An item with no verdict is +identifier, not that nothing happened. An item with no verdict is unanswered, not clean. Say so explicitly; silence will be read as completeness and that is the failure mode this record exists to prevent. @@ -235,7 +202,7 @@ def _no_such_plan(name): "description": "Every recorded fact about one subject, in the order " "the facts took effect, with the actor who asserted " "each and both timestamps. An empty result means " - "nothing was recorded under that identifier — not " + "nothing was recorded under that identifier, not " "that nothing happened.", "inputSchema": { "type": "object", @@ -412,7 +379,7 @@ def main(argv=None): print(f"clew: {len(store[0])} bundles, {len(store[1])} log entries, " f"read-only", file=sys.stderr) if store[2]: - print(f"clew: WARNING — {len(store[2])} sequence conflicts; these " + print(f"clew: WARNING, {len(store[2])} sequence conflicts; these " f"bundles were sealed from different logs. Every answer drawn " f"from the combined history says so.", file=sys.stderr) @@ -426,7 +393,7 @@ def main(argv=None): respond(None, error={"code": -32700, "message": "parse error"}) continue if not isinstance(message, dict): - # Valid JSON, wrong shape — a bare string or list parses fine and + # Valid JSON, wrong shape, a bare string or list parses fine and # then has no .get(). One malformed line must not end a session an # auditor is in the middle of. respond(None, error={"code": -32600, diff --git a/clew/views/report.py b/clew/views/report.py index 23bfc0e..a1df827 100644 --- a/clew/views/report.py +++ b/clew/views/report.py @@ -1,26 +1,12 @@ """ -Clew — one self-contained HTML page for a single impact plan. +One self-contained HTML page for a single impact plan. clew impact --graph g.json --trigger input:reference.dat --html plan.html -The evidence dashboard renders sealed bundles, which is right for an -audit and heavy for the question "what does this change reach, and where -does it run". This renders one plan, straight from `clew impact`, with -the same rules the dashboard follows: - - one file, no scripts, no network, prints legibly, and no generation - timestamp, so the same plan always renders to the same bytes - -It borrows the dashboard's stylesheet and helpers rather than growing a -second set. Two surfaces answering the same question two ways would -eventually disagree, and on that day nobody could say which was wrong. - -WHAT IS NOT KNOWN IS SHOWN --------------------------- -Undetermined verdicts and the limits of the cost figures sit at the top, -in the same weight as everything else. A page that renders gaps in small -grey text below the fold manufactures a clean bill of health out of an -incomplete record. +Same rules as the dashboard: one file, no scripts, no network, prints +legibly, no timestamp. It borrows the dashboard's stylesheet and helpers so +the two cannot disagree. Undetermined verdicts and the limits of the cost +figures sit at the top at full weight. """ import argparse @@ -56,12 +42,9 @@ def possible_actions(item): """ - The actions a task could take once storage is known. - - `possible` maps each candidate action to the rule that would produce - it, and is present precisely when `action` is not. Rendering it as a - blank cell would read as "nothing to do", which is the failure the - plan format guards against. + The actions a task could take once storage is known. `possible` is + present exactly when `action` is not; a blank cell would read as nothing + to do. """ possible = item.get("possible") if isinstance(possible, dict): @@ -156,13 +139,8 @@ def by_target(plan): def by_process(plan): """ - Rolled up by process. Nobody acts on one task at a time, and a flat - list of 183 rows is not a thing anyone reads. - - The target column appears only when something recorded one. An - engine that runs a whole workflow on one machine has no host to - report, and a column of "not recorded" is noise pretending to be - information. + Rolled up by process, since nobody acts on 183 rows one at a time. The + target column appears only when an engine recorded one. """ shown = any(i.get("target") for i in plan["plan"]) @@ -237,7 +215,7 @@ def tasks(plan): if shown: cells.append(f'{esc(i.get("target", ""))}') cells += [f"{esc(i.get('storage') or 'not checked')}", - f'{esc(i.get("reason", ""))}'] + f'{esc(i.get("evidence", i.get("reason", "")))}'] rows += f"{''.join(cells)}" heads = ["Task", "Process"] diff --git a/clew/views/style.py b/clew/views/style.py index 9af9cc3..7dd2455 100644 --- a/clew/views/style.py +++ b/clew/views/style.py @@ -1,40 +1,15 @@ """ -Clew — the QuietFlare report stylesheet. - -Tokens are copied verbatim from quietflare.net, in the same HSL triplet -form and under the same names, so a change on the site is a copy rather -than a translation. - - --accent 25 95% 53% the orange in "Flare" - --foreground 215 28% 17% ink - --primary 222 47% 11% near-black - --steel 215 16% 47% muted text - --border 214 20% 88% - --background 210 40% 98% - --radius .5rem - -LIGHT ONLY, DELIBERATELY ------------------------- -The site is light and these pages match it. Every colour is painted -explicitly rather than inherited, so the page holds its own appearance on -a dark host background instead of borrowing one. - -NO NETWORK, AND THE REAL FACES ------------------------------- -The site loads Inter and Inter Tight from a font host. A report cannot: -one that fetches anything stops opening on a machine with no access, and -these get emailed and archived. So the faces are embedded as woff2 data -URIs instead, which costs about 185 KB and buys a page that looks like -QuietFlare wherever it is opened. - -ORANGE IS BRAND, NEVER STATUS ------------------------------ -The accent sits at hue 25, which is where "warning" normally lives. If -both used it, a reader could not tell "this is QuietFlare" from "this -needs attention". So orange is reserved for identity (wordmark, eyebrow -labels, links, focus) and an unsettled verdict is rendered in steel -rather than amber. That is also truer to what UNDETERMINED means: not -alarming, unanswered. +The QuietFlare report stylesheet. + +Tokens are copied from quietflare.net in the same HSL form and names: accent +25 95% 53%, foreground 215 28% 17%, primary 222 47% 11%, steel 215 16% 47%, +border 214 20% 88%, background 210 40% 98%, radius .5rem. + +Light only, every colour painted explicitly, so the page holds its look on a +dark host. Fonts are embedded as woff2 data URIs, about 185 KB, because a +report that fetches anything stops opening offline and these get emailed and +archived. Orange is identity, never status: an unsettled verdict is rendered +in steel, which is truer to what UNDETERMINED means. """ from clew.views.fonts import FACES diff --git a/docs/adr/0010-built-ins-are-provider-packages.md b/docs/adr/0010-built-ins-are-provider-packages.md new file mode 100644 index 0000000..bfc5745 --- /dev/null +++ b/docs/adr/0010-built-ins-are-provider-packages.md @@ -0,0 +1,76 @@ +# ADR 0010: The built-in adapters and extractors are provider packages + +Status: accepted + +## Context + +Clew publishes two contracts, `Adapter` and `Extractor`, and finds +providers by entry point. A third party ships a package that declares +its modules. Clew's own adapters and extractors were declared the same +way but lived in `clew/domains/` and `clew/extract/`, folders inside the +engine. Two shapes for one thing, and the engine still reached into +them in four places a third party could not: results-tree helpers in +`domains/nfcore.py`, a bookkeeping default there too, `extract/runs.py` +recognising Nextflow and Horus records by name, and `demo.py` importing +`sarek`. + +## Decision + +`clew` is a namespace package. The engine is one distribution and each +provider is another, installing into `clew.provider.`: + +``` +clew/ clew-lineage: graph, ledger, contracts, questions, views, + and extract/ for the command, runs, stitch and digest +providers/ + clew-nextflow/ clew/provider/nextflow/: store, work, rocrate, + NextflowAdapter, sarek, rnaseq, viralrecon + clew-snakemake/ extractor and domain + clew-cromwell/ extractor + clew-horus/ extractor + clew-dnanexus/ extractor + clew-latch/ extractor +``` + +Each has its own `pyproject.toml`, entry points and README, depends on +`clew-lineage`, and is built as a third party would build one: +`from clew.contracts import Adapter, Extractor`, and for an nf-core +adapter `from clew.provider.nextflow import NextflowAdapter`. The engine +depends on none of them and declares extras instead: + +```bash +pip install "clew-lineage[nextflow]" +pip install "clew-lineage[all]" +``` + +The four leaks closed by moving engine-neutral code into the engine and +engine-specific code behind the contract: + +| Was | Now | +|---|---| +| `index_results`, `published_copies` in nfcore | `clew/graph/results.py` | +| `BOOKKEEPING` default in nfcore | `clew/graph/results.py` | +| CSV reader and tag parser in nfcore | `clew/graph/subjects.py`, used by both Nextflow and Snakemake adapters | +| `runs.py` knows Nextflow and Horus | `Extractor.records(path)` and `load(root, run_id)`, optional; `Runs` asks every installed extractor | +| `demo.py` imports sarek | asks `discover(Adapter)` for `sarek` and names the package to install if absent | + +A test holds providers to the same line: a provider may import +`clew.graph`, `clew.contracts` and `clew.extract`, never the questions +and never another provider. + +## Consequences + +The engine can be released without the providers, and a provider +without the engine. The check that a provider works from outside the +engine is no longer a thought experiment: the built-ins are that check. + +A checkout needs all seven distributions installed editable: `make dev`. +The `PYTHON` variable matters on a machine with more than one Python. +PyPI grows from one distribution to seven. + +Each provider's tests live in its own `tests/` with its own fixtures, and +`make test` runs the engine's suite then each provider's. Where a test +exercised an engine feature over one engine's record, `--runs` on a +`.lineage` store or a Horus root, it moved with that record. The engine's +own tests that use the shipped sarek run need `clew-nextflow` installed, +which `make dev` guarantees. diff --git a/docs/adr/0011-trigger-kinds-are-the-adapters-words.md b/docs/adr/0011-trigger-kinds-are-the-adapters-words.md new file mode 100644 index 0000000..9d5ee57 --- /dev/null +++ b/docs/adr/0011-trigger-kinds-are-the-adapters-words.md @@ -0,0 +1,49 @@ +# ADR 0011: Trigger kinds are the adapter's words + +Status: accepted + +## Context + +The domain contract required two methods, `load_subjects` and +`subject_entry_nodes`, and `impact` had `--subject` and `--samplesheet` +flags. All four came from the sarek adapter. A pipeline with no +samplesheet, DNAnexus or Latch or an imaging workflow on Cromwell, had to +implement methods it had nothing to say about, and every user typed a +genomics word the engine had no basis to know. The engine already +resolved `container`, `script`, `process`, `input` and any label without +an adapter; only the owned kind was hardcoded. + +## Decision + +A trigger is `kind:value`. A kind is a small object with a mode, optional +flags of its own, and `resolve(graph, value, args)` returning entry +nodes. The engine ships the four kinds every graph can answer by +contract. A domain declares the rest in its own words: + +```python +triggers = {"patient": SheetKind(column="patient", members="sample")} +``` + +Nothing is required of a domain. Lookup is the adapter's kinds, then the +engine's, then a label key the graph carries; an adapter's kind wins over +the engine's, and `clew providers` marks the shadow. The mode is a closed +enum, trace or remove, because a new verdict would change what +remediation means. The kind names are open, because each is a column in +someone's sheet or a field in someone's LIMS. + +`--subject`, `--donor` and the engine's `--samplesheet` are gone. +`--samplesheet` now exists only where an nf-core kind declared it, and a +kind that can find its inputs from the run declares nothing. The gate +asks a kind for its values instead of reading a sheet itself. + +## Consequences + +The engine's whole vocabulary for what can go wrong is one sentence: a +kind resolves to entry nodes and is traced or removed. The words +inside it belong to providers. Three of the six shipped providers declare +no kinds and are complete. + +Two hooks remain where the engine still holds a judgement that belongs to +a provider: assigning a contribution class from evidence alone, and the +shipped policy table being the default. Both are the same shape as this +decision and are the next two. diff --git a/docs/architecture.md b/docs/architecture.md index 06a142f..10dd9b5 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -23,7 +23,7 @@ flowchart TB end ENG -- extract --> G["One graph
tasks, edges, content digests"] DIG["clew digest
hashes a run once"] --> G - PA --> DOM["domains/
subject → graph nodes"] + PA --> DOM["provider adapters
subject → graph nodes"] G --> Q DOM --> Q subgraph Q["questions/"] @@ -41,14 +41,21 @@ clew/graph/ the one graph: loading, triggers, traversal, contribution classes, digest indexes. Imports nothing else from clew. clew/ledger/ versioned policy, event log, evidence bundles, the gate decision, the query surface, and their commands. -clew/domains/ the layer allowed to know about sarek, samplesheets, donors. -clew/extract/ one extractor per engine, all emitting the same JSON, - plus stitch, digest, and runs, which reads the engine's - record directly. +clew/contracts/ the two provider contracts, Adapter and Extractor, and + how a named subclass registers itself. What a third-party + package imports. Imports only graph. +clew/extract/ the clew extract command, and the engine-neutral tools: + runs, which reads an engine's record through whichever + installed extractor recognises it, stitch, and digest. +providers/ six distributions, one per engine, each installing into + clew.provider. and built as a third party would + build one. clew-nextflow also carries the nf-core + adapters. See ADR 0010. clew/questions/ one module per question asked of the graph: impact, reclaim, drift, gate. clew/views/ dashboard, one page per question, MCP server. -tests/ stdlib unittest. +tests/ the engine's tests. Each provider's are in its own tests/. + All stdlib unittest. ``` Packages import downward only. `graph/` imports nothing from clew, and @@ -57,12 +64,14 @@ lives in `questions/` and imports `clew.graph`, which is also the public API: `load_graph`, `blast_radius`, `classify`, `parse_trigger`, `resolve_trigger`. -The vocabulary boundary is enforced by a grep. `graph/` and `ledger/` must -never mention a sample, a donor, a consent, or a workflow engine. Adding a -new domain, say AI training data with opt-out semantics, means adding a -directory rather than editing the engine. Both rules are -[tests/test_core_boundary.py](../tests/test_core_boundary.py). An unenforced -rule stays true right up until it doesn't. +The vocabulary boundary is enforced by a grep. `graph/`, `ledger/` and +`contracts/` must never mention a sample, a donor, a consent, or a workflow +engine. Adding a new domain, say AI training data with opt-out semantics, +means adding a provider package rather than editing the engine, and the +six shipped providers are held to the same rule: they import the graph, +the contracts and the extract tools, never a question and never each +other. All of it is [tests/test_core_boundary.py](../tests/test_core_boundary.py). +An unenforced rule stays true right up until it doesn't. ## Identity @@ -75,10 +84,13 @@ them; `clew digest` supplies them for runs that were recorded without. ## Tests ```bash -python3 -m unittest discover -s tests +make test ``` -471 tests need nothing installed. Another 25 exercise the log's storage +That runs the engine's suite, then each provider's from its own `tests/` +with its own fixtures. A provider's tests use only what a third party +could: the installed engine and the provider itself. Most tests need +nothing but the checkout installed editable (`make dev`). Another 25 exercise the log's storage behaviour, the role grants, the triggers and concurrent appends, and skip unless you point them at a database you own: @@ -125,7 +137,7 @@ change to any of them is a new ADR that supersedes the old, not an edit. ## Not built A log identity, so bundles from different logs are detected rather than -distinguished. A subject-facing transparency log. Domain adapters beyond +distinguished. A subject-facing transparency log. Adapters beyond nf-core pipelines, though the generic label trigger covers engines that record labels, Horus among them. Storage backends other than a local filesystem for reclaim and digest. diff --git a/docs/evidence.md b/docs/evidence.md index 7d982c1..fc3478c 100644 --- a/docs/evidence.md +++ b/docs/evidence.md @@ -22,7 +22,7 @@ infrastructure defeats that. ``` ok files 6 files, all hashes match ok log 2 entries re-chain to the recorded head (seq 2) - ok policy v2 matches the hash the plan cites + ok policy v1 matches the hash the plan cites ok replay all 57 verdicts recompute identically from the bundled facts and policy ok signature sealed by qa.lead@example.org ``` @@ -43,7 +43,7 @@ item's `possible` map is recomputed, so "one of three" cannot quietly become "one of one". The header's `actions` counts and `tasks_affected` are recomputed from the items. A fact recorded as `null` on any dimension is unverified and evaluated over every value it could take, so a plan with -`terminal: null` cannot replay to `NOTIFY_ONLY`. A fact outside its +`released: null` cannot replay to `NOTIFY_ONLY`. A fact outside its dimension's possible values, `"writable"` for `"WRITABLE"` say, is a discrepancy rather than a silent fall-through. diff --git a/docs/policy.md b/docs/policy.md index 17433ff..4e36c6c 100644 --- a/docs/policy.md +++ b/docs/policy.md @@ -13,7 +13,7 @@ clew rulebook show ``` ``` - R3 exclusive=True, storage=WRITABLE -> DESTROY + R3 scope=exclusive, storage=WRITABLE -> DESTROY Exists only because of this subject and the bytes can be changed. Nothing else needs it, so it goes entirely. ``` @@ -24,54 +24,32 @@ answer" is a checkable sentence. ## Rule order is semantics -First match wins, and an omitted dimension is a wildcard. That is the whole -difference between the two shipped versions. +First match wins, and an omitted dimension is a wildcard. Release is asked +first: a released artifact is `NOTIFY_ONLY` whatever the disk says, because +deleting our copy does not reach the released one. Existence is asked +second, so nothing else is decidable without a storage check, and Clew +withholds rather than guesses. The mode is asked before purge: a separable +part that was corrected (`trace`) is recomputed, R5, while one that was +removed (`remove`) is cut out in place, R6. ```bash -clew rulebook diff v1 v2 +clew impact --graph clew/data/graph5.json --trigger patient:donor_003 \ + --samplesheet clew/data/donors.csv --assertions clew/data/assertions.json ``` ``` -order v1: R1 R2 R3 R4 R5 R6 R7 R8 - v2: R2 R1 R3 R4 R5 R6 R7 R8 - - R2 position 2 -> 1 - - Immutable history: published, or already past a trust boundary. - Terminates remediation, not notification: you cannot unpublish... - + Immutable history: published, or already past a trust boundary. - Asked first, before existence: destroying our copy does not reach - the published or transferred one, so the obligation to disclose - survives the bytes. +POLICY: v1 f1f49f91c8a49f7e + NOTIFY_ONLY (1) c9/023b13 MULTIQC + UNDETERMINED (15) ``` -v1 asked "does it still exist?" before "was it published?", so a published -artifact whose working copy had been deleted came back `ALREADY_GONE`. -Deleting your copy of something does not unpublish it. +## Versions are immutable -The consequence was sharper than it looks. `R1` is the only rule that can -yield `ALREADY_GONE`, so putting it first made every verdict depend on the -storage state. Under v1 nothing was decidable without a disk check. Under v2 -a published artifact resolves without one, because the answer does not -depend on it: - -```bash -clew impact --graph clew/data/graph5.json --samplesheet clew/data/donors.csv \ - --subject donor_003 --assertions clew/data/assertions.json --policy v1 -``` - -``` -POLICY: v1 dbb59de6d85fc0f8 POLICY: v2 e6ba60ffe6763949 - UNDETERMINED (16) NOTIFY_ONLY (1) c9/023b13 MULTIQC - c9/023b13 MULTIQC UNDETERMINED (15) -``` - -## Old versions stay, byte for byte - -`--policy v1` still resolves, and a plan computed in January replays under -the table that produced it rather than under today's. A semantic change is a -new version, never an edit. The shipped hashes are frozen as literals in the -test suite, so editing one fails the build and says to add a version -instead. +A plan cites the table by version and hash, so a plan computed in January +replays under the table that produced it. A semantic change is a new +version, never an edit. The shipped hash is frozen as a literal in the test +suite, so editing the table fails the build and says to add a version +instead. Only `v1` ships today. Adoption is a logged fact. `clew rulebook register` writes a `PolicyAdopted` event carrying the whole table, not a pointer to it. A @@ -105,5 +83,5 @@ sits in the open with a rule id on it. This is the core table, not a customer's policy. It defines what the classes mean, so changing it changes the semantics of every historical plan, which is why it is versioned. Which of a customer's events map to which class, what -counts as published, and what a given withdrawal tier may reach are a -separate object that lives in `domains/`. +counts as released, and what a given trigger may reach are the +adapter's, under `providers/`. diff --git a/docs/providers.md b/docs/providers.md new file mode 100644 index 0000000..5fa0e98 --- /dev/null +++ b/docs/providers.md @@ -0,0 +1,223 @@ +# Providers + +Clew ships the engine and two contracts. A provider is a package that +fills one or both for its own site: a **domain** for a pipeline Clew has +never seen, an **extractor** for an engine Clew cannot read. Both are +found by name and need no change inside Clew. + +```python +from clew.contracts import Adapter, Extractor +``` + +## A domain + +A domain is what a site knows about one pipeline. Clew's core sees a +graph of task ids and nothing else. The domain is where a task named +`BWAMEM1_MEM (SPC-0412)` becomes "specimen SPC-0412 enters here". + +Nothing is required. A domain declares the trigger kinds it understands, +in its own words, and each kind knows how to resolve a value and whether +losing that value is traced or removed. A domain that declares none +still answers the engine's kinds: `container`, `script`, `process`, +`input`, and any label the graph carries. + +| Attribute | What it holds | Default | +|---|---|---| +| `triggers` | `{kind name: Trigger}` | `{}` | +| `load_bearing_inputs` | reference files that are triggers in their own right | `()` | +| `pending()` | triggers the site has recorded and not yet asked | `[]` | + +### The short form: an nf-core pipeline + +nf-core pipelines launch from a CSV samplesheet, and Nextflow names each +task `PROCESS (tag)` with an id from that sheet. `SheetKind`, from the +`clew-nextflow` provider, joins the two. You name the column and the word +your pipeline uses for it. + +```python +from clew.provider.nextflow import NextflowAdapter, SheetKind + +class QbcWgs(NextflowAdapter): + name = "qbc-wgs" # what --pipeline accepts + triggers = {"specimen": SheetKind(column="specimen_id")} # what --trigger accepts + load_bearing_inputs = ("GRCh38_qbc.fa", "qbc_panel_v3.bed") +``` + +```bash +clew impact --graph run.json --pipeline qbc-wgs --trigger specimen:SPC-0412 --samplesheet sheet.csv +``` + +`--samplesheet` is there because `SheetKind` declared it. If one subject +owns several rows, say a patient with a normal and a tumour sample, add +`members="sample"` and a tag naming either resolves to the patient. That +is how sarek declares `patient`. + +### The long form: your own kind + +A kind is a small class. Its mode says what losing a value means, its +flags are whatever it needs, and `resolve` returns entry nodes. A +removal returns every value of the kind, since exclusive and shared are +computed against the others. + +```python +from clew.contracts import Adapter, Trigger, REMOVE + +class LotKind(Trigger): + """Every task that ran while a reagent lot was in use.""" + mode = REMOVE + + def add_arguments(self, parser): + parser.add_argument("--lots", help="JSON: {lot: [task hashes]}") + + def resolve(self, graph, value, args): + # {"LOT-7": ["ab/cdef12", "3f/9a0b44"], ...} + return json.load(open(args.lots)) + + def values(self, args, graph=None): # optional, for the gate + return sorted(json.load(open(args.lots))) + +class QbcLegacy(Adapter): + name = "qbc-legacy" + triggers = {"lot": LotKind()} +``` + +Over-include. A task wrongly listed costs a re-run. A task wrongly +omitted tells someone their data is clean when it is not. + +### Finding inputs without a flag + +A kind that knows where its site keeps things needs no flag. `SheetKind` +takes a `locate` callable that turns the graph into a sheet path, from +the directory the tasks ran in: + +```python +def sheet_beside_the_run(graph): + workdir = next(iter(graph["tasks"].values()))["workdir"] + return str(Path(workdir).parents[2] / "samplesheet.csv") + +triggers = {"specimen": SheetKind(column="specimen_id", locate=sheet_beside_the_run)} +``` + +`--samplesheet` still overrides. A kind you write does the same inside +`resolve`. + +### Running unattended: `pending()` + +Return the triggers your site has recorded and not yet asked about. Each +is a kind and a value, plus who asserted it and when if the source knows. + +```python +class QbcWgs(NextflowAdapter): + ... + def pending(self): + rows = registry_client.withdrawals(status="new") + return [{"kind": "specimen", "value": r.specimen_id, + "asserted_by": r.recorded_by, "date": r.recorded_at} + for r in rows] +``` + +Then the whole question is: + +```bash +clew impact --graph run_42.json --pipeline qbc-wgs +``` + +One plan per trigger, one output file per plan, exit 1 if any answer +failed. A `--trigger` on the command line asks that one question instead +and ignores the queue. See [triggers](triggers.md). + +## An extractor + +An extractor reads one engine's record of a run and returns the graph. +The base owns the parser, the schema check, the summary and `--json-out`. +You add the source flags and the extraction. + +```python +from clew.contracts import Extractor + +class QbcScheduler(Extractor): + name = "qbc-sched" + description = "the QBC scheduler's per-job manifests" + + def add_arguments(self, parser): + parser.add_argument("--manifests", required=True) + + def extract(self, args): + return build_graph(args.manifests) +``` + +`extract` returns `{"tasks": {...}, "edges": [...], "outputs": {...}}`. +The shape is checked before anything is written, and a violation names +the field. Return `None` when the command has already answered, such as a +`--list-runs` listing. Override `summarize(graph, args)` to print +engine-specific lines. + +```bash +clew extract qbc-sched --manifests /jobs --json-out legacy.json +``` + +## Packaging + +A provider installs into Clew's own namespace, `clew.provider.`. +`clew` and `clew.provider` are namespace packages, so your distribution +contributes a directory to them and must not put an `__init__.py` at +either level: + +``` +clew-qbc/ + pyproject.toml + clew/ + provider/ + qbc/ + __init__.py + domain.py class QbcWgs(NextflowAdapter) + extractor.py class QbcScheduler(Extractor) + tests/ + fixtures/ one small real record from your engine + test_extractor.py what it emits passes contract_violations +``` + +One `pyproject.toml`, one entry point per contract you fill. Installing +the package is what delivers them. + +```toml +[project] +name = "clew-qbc" +version = "0.1.0" +dependencies = ["clew-lineage>=0.5", "clew-nextflow>=0.5"] + +[project.entry-points."clew.adapters"] +qbc-wgs = "clew.provider.qbc.domain" + +[project.entry-points."clew.extractors"] +qbc-sched = "clew.provider.qbc.extractor" + +[tool.setuptools.packages.find] +include = ["clew*"] +namespaces = true +``` + +The six providers Clew ships live under `providers/` in its repository +and are built exactly this way. Copy one to start. + +The entry point names a module. Importing it defines the class, and +defining a class with a `name` is what registers it. A class that is +missing a required method fails at definition with the method named. + +While developing, install it editable and every save is live: + +```bash +python3 -m pip install -e ./clew-qbc +``` + +To see what Clew found and where each came from: + +```bash +clew providers +``` + +A provider that failed to import is listed with the error. A module that +imported but defined no named class is listed as registering nothing. + +Nothing leaves the machine. Clew reads the engine's files and your +adapter's answers, and writes a graph and a plan beside them. diff --git a/docs/sources.md b/docs/sources.md index edff9a2..15699b1 100644 --- a/docs/sources.md +++ b/docs/sources.md @@ -54,7 +54,7 @@ engine could not, and every later load merges it back. Nothing the engine already says is copied. `--run` takes a run name, a run-id prefix, or a session-id prefix, as -`extract-store --run` does; a session prefix names a resume chain and its +`clew extract nextflow --run` does; a session prefix names a resume chain and its newest run stands for it. Without `--run` the latest run is read, ordered by the timestamp in the engine's record. A graph directory whose files carry no timestamp is ordered by file modification time, and the command @@ -86,11 +86,11 @@ to run first. Clew has no opinion on where either setting lives. It reads the store the engine writes. ```bash -clew extract-store --store /path/to/.lineage --list-runs +clew extract nextflow --store /path/to/.lineage --list-runs ``` ```bash -clew extract-store --store /path/to/.lineage --run --json-out graph.json +clew extract nextflow --store /path/to/.lineage --run --json-out graph.json ``` The engine is the best witness of what it ran. Inputs are typed, external @@ -108,7 +108,7 @@ paths alone cannot join a run back together. Digests can, and that is what this extractor joins on. ```bash -clew extract-horus --run-dir ~/.horus-lineage// --json-out graph.json +clew extract horus --run-dir /.horus-lineage// --json-out graph.json ``` Skipped tasks are recorded with their digests, so a cached run gives the @@ -125,7 +125,7 @@ on file ID, so two analyses stitch with no path matching, even across projects. ```bash -clew extract-dnanexus --analysis analysis-xxxx --json-out graph.json +clew extract dnanexus --analysis analysis-xxxx --json-out graph.json ``` The token comes from `DX_SECURITY_CONTEXT`, which `dx login` sets, or from @@ -158,7 +158,7 @@ shared path like two Nextflow runs do. The workflow's commit hash and image hash identify the code and the environment. ```bash -clew extract-latch --execution --json-out graph.json +clew extract latch --execution --json-out graph.json ``` The token is the one `latch login` stores, or `--token`. The extractor @@ -184,14 +184,14 @@ as paths, so an input that is another call's output is an edge, and a path no call produced came from outside. ```bash -clew extract-cromwell --metadata metadata.json --json-out graph.json +clew extract cromwell --metadata metadata.json --json-out graph.json ``` The file is what `cromwell run -m metadata.json` writes. From a server, fetch it yourself with subworkflows expanded, or let Clew do it: ```bash -clew extract-cromwell --server http://localhost:8000 --workflow --json-out graph.json +clew extract cromwell --server http://localhost:8000 --workflow --json-out graph.json ``` `--token` sends a bearer token for a server behind auth. The extractor @@ -226,7 +226,7 @@ it decides what to rerun. That memory is lineage. Nothing changes in the workflow. ```bash -clew extract-snakemake --workdir /path/to/workflow --json-out graph.json +clew extract snakemake --workdir /path/to/workflow --json-out graph.json ``` Both persistence backends are read: the JSON files under @@ -275,7 +275,7 @@ Workflow Run RO-Crate. Labs that publish crates for journals or archives already have lineage on disk. ```bash -clew extract-crate --crate ro-crate-metadata.json --json-out graph.json +clew extract ro-crate --crate ro-crate-metadata.json --json-out graph.json ``` A crate records what ran, not how to run it again. There is no script, no @@ -296,7 +296,7 @@ symlinks, and those symlinks record the whole history of the run. No pipeline change, any Nextflow version: ```bash -clew extract-work --jsonl /path/to/weblog/.jsonl --work /path/to/work --json-out graph.json +clew extract nextflow-work --jsonl /path/to/weblog/.jsonl --work /path/to/work --json-out graph.json ``` Do this during or right after the run. `nextflow clean` removes the diff --git a/docs/storage.md b/docs/storage.md index 6ab2c20..86510d3 100644 --- a/docs/storage.md +++ b/docs/storage.md @@ -14,8 +14,8 @@ someone emailed over. So Clew does not check unless you tell it where to look: ```bash -clew impact --graph graph.json --samplesheet samplesheet.csv \ - --subject donor_003 --work-root /path/to/work +clew impact --graph graph.json --trigger patient:donor_003 \ + --samplesheet samplesheet.csv --work-root /path/to/work ``` Each extractor records where under the engine's root a task ran, as @@ -33,7 +33,7 @@ nothing is checked. Both print one warning on stderr. Without `--work-root`, storage is unverified and any verdict that depends on it comes back `UNDETERMINED` rather than guessed. Verdicts that hold -whatever the disk says are still returned. Under policy v2 a published +whatever the disk says are still returned. A released artifact is `NOTIFY_ONLY` either way, and that is an answer, not a guess: ``` diff --git a/docs/triggers.md b/docs/triggers.md index 3bf5619..a96d854 100644 --- a/docs/triggers.md +++ b/docs/triggers.md @@ -1,28 +1,49 @@ # Triggers -Clew has no hardcoded scenarios. Every trigger combines two independent -choices, a selector and a mode, and the familiar stories are named cells in -that grid. - -## The selector: where does the problem enter the graph? - -| Selector | Flag | Entry nodes | -|---|---|---| -| subject | `--subject X` | every task attributed to one sample, donor or batch | -| container | `--container Y` | every task that ran in a matching container | -| external input | `--input Z` | every task that consumed that outside file | -| generic | `--trigger kind:value` | see below | - -`--donor` is the former name of `--subject` and still works. - -The generic form takes `kind:value`, for example `container:gatk4`, -`script:prep.py`, `input:genome.fa` or `subject:batch_017`. An unknown kind -is read as a label key, so a graph whose tasks carry labels such as -`{tissue: liver}` answers `--trigger tissue:liver` with no adapter and no -new flag. Horus records carry labels natively. `subject:X` with -`--samplesheet` is the same as `--subject X`: nf-core tasks carry no -labels, so the samplesheet is what resolves one. Without a samplesheet it -reads the graph's `subject` labels, and says so when there are none. +A trigger is `kind:value`. The kind says where the problem enters the +graph and whether it is traced or removed. The value says which one. + +```bash +clew impact --graph g.json --trigger container:gatk4 +clew impact --graph g.json --trigger patient:donor_003 --samplesheet sheet.csv +clew impact --graph g.json --trigger patient --samplesheet sheet.csv # every patient +``` + +`--container X` and `--input X` are short for the two engine kinds people +type most. + +## Where a kind comes from + +The engine ships four kinds, one per field every graph carries by +contract. Any other word is looked up in this order: + +1. The pipeline's adapter, if it declares a kind by that name. +2. The engine's four: `container`, `script`, `process`, `input`. +3. A label key the graph carries on its tasks or edges. + +| Kind | Declared by | Entry nodes | Mode | +|---|---|---|---| +| `container` | engine | every task whose image matches | trace | +| `script`, `process` | engine | every task whose field contains the value | trace | +| `input` | engine | every task that consumed that outside file | trace | +| any label key | engine | every task or artifact carrying `labels[key] == value` | trace | +| `patient` | the sarek adapter | every task tagged with the patient or one of its samples | remove | +| `sample` | rnaseq, viralrecon, snakemake | every task tagged with the sample | remove | +| yours | your adapter | whatever your code says | your choice | + +The engine never sees the words `patient` or `sample`. It asks the +adapter, and the adapter's kind resolves the value with whatever it needs, +declaring its own flags. `--samplesheet` exists because the nf-core kinds +put it there; on a pipeline with no samplesheet it does not appear. A kind +can also find its inputs from the run itself, and then no flag is typed. + +An unknown kind is refused by name, and the message lists the kinds the +adapter declares, the engine's four, and whether the graph carries such a +label. `clew providers` lists every adapter's kinds and marks any that +shadow an engine kind, since an adapter's kind wins over the engine's. + +Horus records carry labels natively, so `--trigger site:north` works on a +Horus graph with no adapter at all. ### How a container is matched @@ -41,55 +62,54 @@ name, so `samtools` finds `bwa_htslib_samtools`. When the needle carries a version and the image carries none, the task matches on name alone and the output says so, under `TRIGGER NOTES` and in -the plan's caveats. On the shipped sarek run `samtools:1.21` reaches all -26 samtools tasks, 11 of them on name only, where substring matching found -15 and reported 46 affected instead of 72. +the plan's caveats. ### How an input is matched -`--input genome.fasta` matches the exact basename, plus companions named +`input:genome.fasta` matches the exact basename, plus companions named `genome.fasta.`, such as `genome.fasta.fai`, since an index is regenerated with the file it belongs to. A directory input matches by its own basename. The companions reached are listed under `TRIGGER NOTES`. ## The mode: what kind of wrong is it? -Two modes exist, and they are not interchangeable. +`remove` means the source is withdrawn. Ownership matters: an artifact +that exists only because of this value has nothing left to serve, so it +can be destroyed. A removal kind resolves every value of its kind, not +only the one asked about, because exclusive and shared are computed +against the others. -`remove` means the source must be taken out, as in a withdrawal. Ownership -matters. An artifact that exists only because of this subject has nothing -left to serve, so it can be destroyed. +`trace` means follow what the value touched and leave everything in +place, as with contamination, a tool defect or a stale reference. A +removal traces too; the difference is only what may happen at the end. +Under trace nothing is destroyed and the worst verdict is quarantine. -`distrust` means the data is suspect but still wanted, as with -contamination, a tool defect or a stale reference. Nothing is destroyed. The -worst verdict is quarantine, because you will want these artifacts again -once the cause is fixed. +The kind declares its default. `--mode` overrides it in one direction: a +removal kind can be traced (contamination of one patient is +`--trigger patient:X --mode trace`), but a trace kind cannot be asked +as a removal, because nothing is owned. ## The stories, mapped -| Story | Selector | Mode | +| Story | Kind | Mode | |---|---|---| -| Reference or annotation update | external input | distrust | -| Tool or container defect | container | distrust | -| Sample contamination, swap, QC failure | subject | distrust | -| Primer scheme correction | external input | distrust | -| Consent withdrawal | subject | remove | -| Upstream dataset retraction | external input | remove, not yet supported | +| Reference or annotation update | `input` | trace | +| Tool or container defect | `container` | trace | +| Contamination, swap, QC failure | the adapter's owning kind | trace | +| Primer scheme correction | `input` | trace | +| Consent withdrawal | the adapter's owning kind | remove | +| Upstream dataset retraction | `input` | remove, not yet supported | Retraction is unsupported because removal needs an owner, and computing what exists only because of one input needs multi-root traversal. -Defaults preserve the common cases. `--subject` implies remove, the others -imply distrust, and `--mode` overrides. Contamination is -`--subject X --mode distrust`. - ## Extending it -Selectors are the extension point. A selector is anything that can name a -set of entry nodes, and the engine only ever sees the set. Candidates: one -exact artifact by checksum, every task in a time window for a bad reagent -lot, a facility, or an edge kind such as everything calibrated against a -control rather than merely derived from it. +A kind is a small class: a mode, optional flags, and `resolve(graph, +value, args)` returning `{id: [entry nodes]}`. Candidates a site might +write: one exact artifact by checksum, every task in a time window for a +bad reagent lot, a facility, or everything calibrated against a control +rather than merely derived from it. [Providers](providers.md) shows one. Modes are the closed part. A new verdict would change what remediation means, so a new mode is a design decision, not a plugin. diff --git a/examples/clew-gate.yml b/examples/clew-gate.yml index 4f64e4e..275118d 100644 --- a/examples/clew-gate.yml +++ b/examples/clew-gate.yml @@ -78,7 +78,7 @@ jobs: # steps: # - run: | # clew impact --pipeline sarek --graph graph.json \ - # --samplesheet samplesheet.csv --donor "$SUBJECT" \ + # --trigger "patient:$SUBJECT" --samplesheet samplesheet.csv \ # --work-root work/ --json plan.json # clew evidence build --out clew-evidence/ \ # --plan plan.json --input graph.json --input samplesheet.csv diff --git a/providers/clew-cromwell/README.md b/providers/clew-cromwell/README.md new file mode 100644 index 0000000..7859ccc --- /dev/null +++ b/providers/clew-cromwell/README.md @@ -0,0 +1,11 @@ +# clew-cromwell + +Clew provider for Cromwell workflow metadata, for WDL and Terra. + +```bash +pip install clew-cromwell +``` + +Registers under the entry-point groups Clew discovers. `clew providers` +lists it. Built exactly as a third-party provider would be: see +[providers](../../docs/providers.md) in the engine's docs. diff --git a/clew/domains/__init__.py b/providers/clew-cromwell/clew/provider/cromwell/__init__.py similarity index 100% rename from clew/domains/__init__.py rename to providers/clew-cromwell/clew/provider/cromwell/__init__.py diff --git a/clew/extract/cromwell.py b/providers/clew-cromwell/clew/provider/cromwell/extractor_metadata.py similarity index 75% rename from clew/extract/cromwell.py rename to providers/clew-cromwell/clew/provider/cromwell/extractor_metadata.py index 5bdf74d..93cbfed 100644 --- a/clew/extract/cromwell.py +++ b/providers/clew-cromwell/clew/provider/cromwell/extractor_metadata.py @@ -5,7 +5,6 @@ Details and limits: docs/sources.md, "Cromwell". """ -import argparse import json import sys import urllib.parse @@ -14,6 +13,7 @@ from pathlib import Path from clew.graph.graph import relative_to, task_status +from clew.contracts import Extractor EXTERNAL = "EXTERNAL" SCHEMES = ("gs://", "s3://", "drs://", "http://", "https://", "az://", "file://") @@ -166,46 +166,47 @@ def fetch(server, workflow_id, token=None): return json.loads(response.read()) -def main(argv=None): - parser = argparse.ArgumentParser( - description="Build a Clew graph from Cromwell workflow metadata.") - source = parser.add_mutually_exclusive_group(required=True) - source.add_argument("--metadata", - help="metadata JSON, from `cromwell run -m` or the API") - source.add_argument("--server", help="Cromwell server URL, with --workflow") - parser.add_argument("--workflow", help="workflow id to fetch from --server") - parser.add_argument("--token", help="bearer token for a server behind auth") - parser.add_argument("--json-out", help="path to write the graph as JSON") - args = parser.parse_args(argv) - - if args.metadata: - metadata = load_metadata(args.metadata) - else: - if not args.workflow: - print("clew: --server needs --workflow ", file=sys.stderr) - return 2 - metadata = fetch(args.server, args.workflow, args.token) - - graph = extract(metadata) - external = [e for e in graph["edges"] if e["producer"] == EXTERNAL] - cached = [t for t in graph["tasks"].values() if t.get("cached")] - unexpanded = [t for t in graph["tasks"].values() - if t.get("unexpanded_subworkflow")] - - print(f"workflow : {metadata.get('workflowName', '?')} " - f"{metadata.get('id', '')} {metadata.get('status', '')}") - print(f"calls : {len(graph['tasks'])}") - print(f" cache hits : {len(cached)}") - print(f"input files (edges): {len(graph['edges'])}") - print(f" external inputs : {len(external)}") - if unexpanded: - print(f" UNEXPANDED subworkflows: {len(unexpanded)} " - f"(fetch with expandSubWorkflows=true)") - - if args.json_out: - Path(args.json_out).write_text(json.dumps(graph, indent=2)) - print(f"\nwrote {args.json_out}") - return 0 +class Cromwell(Extractor): + name = "cromwell" + description = "Cromwell workflow metadata, for WDL and Terra" + + def add_arguments(self, parser): + source = parser.add_mutually_exclusive_group(required=True) + source.add_argument("--metadata", + help="metadata JSON, from `cromwell run -m` or the API") + source.add_argument("--server", help="Cromwell server URL, with --workflow") + parser.add_argument("--workflow", help="workflow id to fetch from --server") + parser.add_argument("--token", help="bearer token for a server behind auth") + + def extract(self, args): + if args.metadata: + self.metadata = load_metadata(args.metadata) + else: + if not args.workflow: + print("clew: --server needs --workflow ", file=sys.stderr) + raise SystemExit(2) + self.metadata = fetch(args.server, args.workflow, args.token) + return extract(self.metadata) + + def summarize(self, graph, args): + metadata = self.metadata + external = [e for e in graph["edges"] if e["producer"] == EXTERNAL] + cached = [t for t in graph["tasks"].values() if t.get("cached")] + unexpanded = [t for t in graph["tasks"].values() + if t.get("unexpanded_subworkflow")] + print(f"workflow : {metadata.get('workflowName', '?')} " + f"{metadata.get('id', '')} {metadata.get('status', '')}") + print(f"calls : {len(graph['tasks'])}") + print(f" cache hits : {len(cached)}") + print(f"input files (edges): {len(graph['edges'])}") + print(f" external inputs : {len(external)}") + if unexpanded: + print(f" UNEXPANDED subworkflows: {len(unexpanded)} " + f"(fetch with expandSubWorkflows=true)") + self.coverage(graph) + + +main = Cromwell.main if __name__ == "__main__": diff --git a/providers/clew-cromwell/pyproject.toml b/providers/clew-cromwell/pyproject.toml new file mode 100644 index 0000000..013f73d --- /dev/null +++ b/providers/clew-cromwell/pyproject.toml @@ -0,0 +1,21 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "clew-cromwell" +version = "0.4.0" +description = "Clew provider for Cromwell workflow metadata, for WDL and Terra" +readme = "README.md" +requires-python = ">=3.9" +license = { text = "AGPL-3.0-only" } +authors = [{ name = "QuietFlare" }] +dependencies = ["clew-lineage>=0.4"] + +[project.entry-points."clew.extractors"] +cromwell = "clew.provider.cromwell.extractor_metadata" + +[tool.setuptools.packages.find] +# Installs into the clew.provider namespace; no __init__.py above the provider. +include = ["clew*"] +namespaces = true diff --git a/tests/fixtures/cromwell/diamond.json b/providers/clew-cromwell/tests/fixtures/cromwell/diamond.json similarity index 100% rename from tests/fixtures/cromwell/diamond.json rename to providers/clew-cromwell/tests/fixtures/cromwell/diamond.json diff --git a/tests/fixtures/cromwell/outer.json b/providers/clew-cromwell/tests/fixtures/cromwell/outer.json similarity index 100% rename from tests/fixtures/cromwell/outer.json rename to providers/clew-cromwell/tests/fixtures/cromwell/outer.json diff --git a/tests/test_cromwell.py b/providers/clew-cromwell/tests/test_cromwell.py similarity index 97% rename from tests/test_cromwell.py rename to providers/clew-cromwell/tests/test_cromwell.py index 6f5192c..24ab9e5 100644 --- a/tests/test_cromwell.py +++ b/providers/clew-cromwell/tests/test_cromwell.py @@ -22,9 +22,9 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.extract import cromwell as cw +from clew.provider.cromwell import extractor_metadata as cw from clew.graph import blast_radius as core from clew.graph.graph import contract_violations, external_input_entry_nodes @@ -236,7 +236,9 @@ def test_metadata_file_writes_a_graph(self): def test_server_without_workflow_fails_cleanly(self): with contextlib.redirect_stderr(io.StringIO()): - self.assertEqual(cw.main(["--server", "http://localhost:8000"]), 2) + with self.assertRaises(SystemExit) as stop: + cw.main(["--server", "http://localhost:8000"]) + self.assertEqual(stop.exception.code, 2) if __name__ == "__main__": diff --git a/providers/clew-cromwell/tests/test_cromwell_contract.py b/providers/clew-cromwell/tests/test_cromwell_contract.py new file mode 100644 index 0000000..e7f1243 --- /dev/null +++ b/providers/clew-cromwell/tests/test_cromwell_contract.py @@ -0,0 +1,30 @@ +"""What this extractor emits is the one graph every question reads.""" + +import unittest +from pathlib import Path + +from clew.graph.graph import STATUSES, contract_violations +from clew.provider.cromwell import extractor_metadata as cw + +FIXTURES = Path(__file__).resolve().parent / "fixtures" / "cromwell" + + +class Contract(unittest.TestCase): + def graphs(self): + yield "diamond", cw.extract(cw.load_metadata(FIXTURES / "diamond.json")) + yield "subworkflow", cw.extract(cw.load_metadata(FIXTURES / "outer.json")) + + def test_conforms(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + self.assertEqual(contract_violations(graph), []) + + def test_statuses_are_in_the_vocabulary(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + for task in graph["tasks"].values(): + self.assertIn(task["status"], STATUSES) + + +if __name__ == "__main__": + unittest.main() diff --git a/providers/clew-dnanexus/README.md b/providers/clew-dnanexus/README.md new file mode 100644 index 0000000..3a76816 --- /dev/null +++ b/providers/clew-dnanexus/README.md @@ -0,0 +1,11 @@ +# clew-dnanexus + +Clew provider for DNAnexus analyses, from saved records or the API. + +```bash +pip install clew-dnanexus +``` + +Registers under the entry-point groups Clew discovers. `clew providers` +lists it. Built exactly as a third-party provider would be: see +[providers](../../docs/providers.md) in the engine's docs. diff --git a/providers/clew-dnanexus/clew/provider/dnanexus/__init__.py b/providers/clew-dnanexus/clew/provider/dnanexus/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clew/extract/dnanexus.py b/providers/clew-dnanexus/clew/provider/dnanexus/extractor_describe.py similarity index 68% rename from clew/extract/dnanexus.py rename to providers/clew-dnanexus/clew/provider/dnanexus/extractor_describe.py index 8578ad0..2234dc3 100644 --- a/clew/extract/dnanexus.py +++ b/providers/clew-dnanexus/clew/provider/dnanexus/extractor_describe.py @@ -1,37 +1,20 @@ """ -Clew: lineage adapter for DNAnexus analyses. - -DNAnexus records what Clew needs on the platform itself. A job describe -lists its inputs and outputs as file IDs, a file describe names the job -that created it, and file IDs are immutable and survive cloning between -projects. So edges join on file ID, and a run stitches to another run -with no path matching at all. - -Two ways in. `--records DIR` reads describe output saved as JSON -(jobs.json and files.json), which is how the fixtures and tests work. -`--analysis ID` fetches the same records over the API with a token, using -nothing beyond the standard library. The API path has not yet been run -against a live analysis; the record shapes come from the DNAnexus API -documentation. - -An input given as another job's output field, or as an analysis stage, -is an edge to that job when it belongs to the analysis. One that names -no job here is counted in the graph's `coverage` rather than dropped. - -The task hash is the job ID. Status is the job state, upper-cased. -The container is "@", so a trigger -like --container gatk matches by name. Two optional keys carry what -DNAnexus knows and Nextflow does not: `price`, the job's total price -when the caller has billing access, and `duration_s`, wall time between -startedRunning and stoppedRunning. - -Nextflow pipelines on DNAnexus run as a head job plus one subjob per -process. Whether those subjobs expose per-process files as platform -file IDs is unverified; if they do not, use the Nextflow lineage store -written to the project instead. +DNAnexus analyses, read into the graph. + +A job describe lists inputs and outputs as file IDs, a file describe names +the job that created it, and file IDs survive cloning between projects, so +edges join on file ID. --records DIR reads saved describe output (jobs.json, +files.json). --analysis ID fetches the same over the API with a token, +standard library only, and has not yet met a live analysis. + +The task hash is the job ID, status the job state, container +"@", with optional price and duration_s. An +input naming a job outside the analysis counts in coverage rather than being +dropped. Whether Nextflow subjobs on DNAnexus expose per-process files as +platform IDs is unverified; if not, use the lineage store written to the +project. """ -import argparse import json import os import sys @@ -39,6 +22,7 @@ from pathlib import Path from clew.graph.graph import EXTERNAL, task_status +from clew.contracts import Extractor API = "https://api.dnanexus.com" @@ -69,12 +53,9 @@ def link_ids(value): def job_refs(value): """ - Every job-based or stage reference in a job input, as (kind, id, field). - - A job launched inside an analysis is often given another job's output - field rather than a file ID, and a finished job's describe may still - show it that way. These are edges to a job, not to a file, and they - used to yield nothing. + Every job or stage reference in a job input, as (kind, id, field). An + input given as another job's output field is an edge to that job, not to + a file. """ if isinstance(value, dict): link = value.get("$dnanexus_link") @@ -136,10 +117,10 @@ def extract(records): "workdir": "", } if isinstance(job.get("totalPrice"), (int, float)): - task["price"] = job["totalPrice"] + task.setdefault("metrics", {})["price"] = job["totalPrice"] duration = duration_of(job) if duration is not None: - task["duration_s"] = duration + task.setdefault("metrics", {})["duration_s"] = duration tasks[job_id] = task views = input_views(job) @@ -237,43 +218,37 @@ def token_from_env(): return os.environ.get("DX_API_TOKEN") -def main(argv=None): - parser = argparse.ArgumentParser( - description="Build a Clew graph from a DNAnexus analysis.") - source = parser.add_mutually_exclusive_group(required=True) - source.add_argument("--records", help="directory with jobs.json and files.json") - source.add_argument("--analysis", help="analysis-xxxx to fetch over the API") - parser.add_argument("--token", help="API token; default DX_SECURITY_CONTEXT") - parser.add_argument("--json-out", help="path to write the graph as JSON") - args = parser.parse_args(argv) +class DNAnexus(Extractor): + name = "dnanexus" + description = "a DNAnexus analysis, from saved records or the API" + + def add_arguments(self, parser): + source = parser.add_mutually_exclusive_group(required=True) + source.add_argument("--records", help="directory with jobs.json and files.json") + source.add_argument("--analysis", help="analysis-xxxx to fetch over the API") + parser.add_argument("--token", help="API token; default DX_SECURITY_CONTEXT") - if args.records: - records = load_records(args.records) - else: + def extract(self, args): + if args.records: + return extract(load_records(args.records)) token = args.token or token_from_env() if not token: print("clew: no API token; pass --token or log in with dx", file=sys.stderr) - return 2 - records = fetch(args.analysis, token) - - graph = extract(records) - external = [e for e in graph["edges"] if e["producer"] == EXTERNAL] - priced = [t for t in graph["tasks"].values() if "price" in t] - - print(f"jobs : {len(graph['tasks'])}") - print(f"input files (edges): {len(graph['edges'])}") - print(f" external inputs : {len(external)}") - if priced: - print(f"total price : {sum(t['price'] for t in priced):.2f}") - if graph.get("coverage"): - print("\n=== what this graph does not cover ===") - for note in graph["coverage"]: - print(f" - {note}") - - if args.json_out: - Path(args.json_out).write_text(json.dumps(graph, indent=2)) - print(f"\nwrote {args.json_out}") - return 0 + raise SystemExit(2) + return extract(fetch(args.analysis, token)) + + def summarize(self, graph, args): + external = [e for e in graph["edges"] if e["producer"] == EXTERNAL] + priced = [t for t in graph["tasks"].values() if "price" in t.get("metrics", {})] + print(f"jobs : {len(graph['tasks'])}") + print(f"input files (edges): {len(graph['edges'])}") + print(f" external inputs : {len(external)}") + if priced: + print(f"total price : {sum(t['metrics']['price'] for t in priced):.2f}") + self.coverage(graph) + + +main = DNAnexus.main if __name__ == "__main__": diff --git a/providers/clew-dnanexus/pyproject.toml b/providers/clew-dnanexus/pyproject.toml new file mode 100644 index 0000000..37e2d57 --- /dev/null +++ b/providers/clew-dnanexus/pyproject.toml @@ -0,0 +1,21 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "clew-dnanexus" +version = "0.4.0" +description = "Clew provider for DNAnexus analyses, from saved records or the API" +readme = "README.md" +requires-python = ">=3.9" +license = { text = "AGPL-3.0-only" } +authors = [{ name = "QuietFlare" }] +dependencies = ["clew-lineage>=0.4"] + +[project.entry-points."clew.extractors"] +dnanexus = "clew.provider.dnanexus.extractor_describe" + +[tool.setuptools.packages.find] +# Installs into the clew.provider namespace; no __init__.py above the provider. +include = ["clew*"] +namespaces = true diff --git a/tests/fixtures/dnanexus/files.json b/providers/clew-dnanexus/tests/fixtures/dnanexus/files.json similarity index 100% rename from tests/fixtures/dnanexus/files.json rename to providers/clew-dnanexus/tests/fixtures/dnanexus/files.json diff --git a/tests/fixtures/dnanexus/jobs.json b/providers/clew-dnanexus/tests/fixtures/dnanexus/jobs.json similarity index 100% rename from tests/fixtures/dnanexus/jobs.json rename to providers/clew-dnanexus/tests/fixtures/dnanexus/jobs.json diff --git a/tests/test_dnanexus.py b/providers/clew-dnanexus/tests/test_dnanexus.py similarity index 92% rename from tests/test_dnanexus.py rename to providers/clew-dnanexus/tests/test_dnanexus.py index 438a6b1..798764b 100644 --- a/tests/test_dnanexus.py +++ b/providers/clew-dnanexus/tests/test_dnanexus.py @@ -11,9 +11,9 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.extract import dnanexus as dx +from clew.provider.dnanexus import extractor_describe as dx from clew.graph import blast_radius as core from clew.graph.graph import contract_violations, external_input_entry_nodes @@ -81,10 +81,10 @@ def test_container_names_the_executable(self): "gatk_haplotypecaller@app-gatk") def test_price_and_duration_when_present(self): - self.assertEqual(self.graph["tasks"][CALL]["price"], 0.35) - self.assertEqual(self.graph["tasks"][CALL]["duration_s"], 1800) - self.assertNotIn("price", self.graph["tasks"][REPORT]) - self.assertEqual(self.graph["tasks"][REPORT]["duration_s"], 60) + self.assertEqual(self.graph["tasks"][CALL]["metrics"]["price"], 0.35) + self.assertEqual(self.graph["tasks"][CALL]["metrics"]["duration_s"], 1800) + self.assertNotIn("price", self.graph["tasks"][REPORT]["metrics"]) + self.assertEqual(self.graph["tasks"][REPORT]["metrics"]["duration_s"], 60) class Triggers(unittest.TestCase): @@ -182,7 +182,9 @@ def test_analysis_without_token_fails_cleanly(self): saved = {k: os.environ.pop(k, None) for k in ("DX_SECURITY_CONTEXT", "DX_API_TOKEN")} try: - self.assertEqual(dx.main(["--analysis", "analysis-A"]), 2) + with self.assertRaises(SystemExit) as stop: + dx.main(["--analysis", "analysis-A"]) + self.assertEqual(stop.exception.code, 2) finally: for k, v in saved.items(): if v is not None: diff --git a/providers/clew-dnanexus/tests/test_dnanexus_contract.py b/providers/clew-dnanexus/tests/test_dnanexus_contract.py new file mode 100644 index 0000000..c4bd15f --- /dev/null +++ b/providers/clew-dnanexus/tests/test_dnanexus_contract.py @@ -0,0 +1,29 @@ +"""What this extractor emits is the one graph every question reads.""" + +import unittest +from pathlib import Path + +from clew.graph.graph import STATUSES, contract_violations +from clew.provider.dnanexus import extractor_describe as dx + +FIXTURES = Path(__file__).resolve().parent / "fixtures" / "dnanexus" + + +class Contract(unittest.TestCase): + def graphs(self): + yield "dnanexus", dx.extract(dx.load_records(FIXTURES)) + + def test_conforms(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + self.assertEqual(contract_violations(graph), []) + + def test_statuses_are_in_the_vocabulary(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + for task in graph["tasks"].values(): + self.assertIn(task["status"], STATUSES) + + +if __name__ == "__main__": + unittest.main() diff --git a/providers/clew-horus/README.md b/providers/clew-horus/README.md new file mode 100644 index 0000000..4732ed9 --- /dev/null +++ b/providers/clew-horus/README.md @@ -0,0 +1,11 @@ +# clew-horus + +Clew provider for horus-lineage run directories. + +```bash +pip install clew-horus +``` + +Registers under the entry-point groups Clew discovers. `clew providers` +lists it. Built exactly as a third-party provider would be: see +[providers](../../docs/providers.md) in the engine's docs. diff --git a/providers/clew-horus/clew/provider/horus/__init__.py b/providers/clew-horus/clew/provider/horus/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/providers/clew-horus/clew/provider/horus/adapter_vina_docking.py b/providers/clew-horus/clew/provider/horus/adapter_vina_docking.py new file mode 100644 index 0000000..fd81527 --- /dev/null +++ b/providers/clew-horus/clew/provider/horus/adapter_vina_docking.py @@ -0,0 +1,75 @@ +""" +Pantheon W-02, AutoDock Vina docking on Horus: prep -> dock -> summary. + +Every ligand enters at prep through the library file, and every output is +per-ligand: a PDBQT in the inputs archive, a pose file in the docking +archive, rows in the two CSVs. Withdrawing a ligand is a removal whose +artifacts are all separable. +""" + +from pathlib import Path + +from clew.contracts import REMOVE, Adapter, Trigger +from clew.graph.contribution import SEPARABLE +from clew.graph.graph import EXTERNAL + +DEFAULT_LIBRARY = "ligands.smi" +PER_LIGAND_STEPS = ("prep", "dock", "summary") + + +def read_library(path): + """{name: []} from a SMILES file, one 'SMILES name' per line.""" + names = {} + for line in Path(path).read_text().splitlines(): + parts = line.split() + if len(parts) >= 2 and not line.startswith("#"): + names.setdefault(parts[1], []) + return names + + +class LigandKind(Trigger): + mode = REMOVE + + def add_arguments(self, parser): + parser.add_argument("--ligands", metavar="SMI", + help=f"the ligand library the run docked; default {DEFAULT_LIBRARY} beside the run") + + def library(self, args): + path = getattr(args, "ligands", None) + if not path: + raise SystemExit("--ligands is required: ligand names come from the library's second column") + return path + + def ids(self, path): + return read_library(path) + + def entries(self, graph, ids, library_name=DEFAULT_LIBRARY): + # The library is one external file consumed by prep, so every ligand + # enters at the same task. Nothing is exclusive; the class hook is + # what makes a single ligand removable. + entry = sorted(e["consumer"] for e in graph["edges"] + if e["producer"] == EXTERNAL and e["filename"] == library_name) + return {name: list(entry) for name in ids} + + def resolve(self, graph, value, args): + path = self.library(args) + entry = self.entries(graph, self.ids(path), Path(path).name) + if value is not None and value not in entry: + raise SystemExit(f"unknown ligand {value!r}; the library names: {', '.join(sorted(entry))}") + return entry + + def values(self, args, graph=None): + return sorted(self.ids(self.library(args))) + + +class VinaDocking(Adapter): + name = "vina-docking" + triggers = {"ligand": LigandKind()} + load_bearing_inputs = ("receptor.pdb",) + + def contribution(self, graph, task_hash, kind): + # prep, dock and summary each write one entry per ligand, so one + # ligand's contribution can be dropped in place. A defect in the step + # itself is not per-ligand, so any other kind keeps the engine's answer. + process = graph["tasks"].get(task_hash, {}).get("process") + return SEPARABLE if kind == "ligand" and process in PER_LIGAND_STEPS else None diff --git a/clew/extract/horus.py b/providers/clew-horus/clew/provider/horus/extractor_lineage.py similarity index 67% rename from clew/extract/horus.py rename to providers/clew-horus/clew/provider/horus/extractor_lineage.py index e1b258f..ce0a149 100644 --- a/clew/extract/horus.py +++ b/providers/clew-horus/clew/provider/horus/extractor_lineage.py @@ -1,62 +1,26 @@ """ -Clew — lineage adapter for horus-lineage run directories. - -WHY THIS EXISTS ---------------- -The other three adapters read Nextflow. Horus is a different engine with a -property Nextflow does not have: every task can run on a different machine, -so a run's outputs are scattered across a laptop, a cluster and whatever -else the workflow named. Paths alone cannot join that back together. - -horus-lineage records a content digest for every input and output, computed -on the machine holding the bytes. That is what this adapter joins on, so a -graph closes across machines the same way it closes on one. - -THE RUN DIRECTORY, AS horus-lineage WRITES IT ---------------------------------------------- - run.json the plan: run id, workflow, timings, final status - definition.json the projected workflow: tasks and declared edges - ..json one record per task - records.jsonl the same records, one per line, when merged - -Both record layouts are read. A task record carries status, the resolved -command, environment and code digests, and every input and output with its -sha256. - -WHAT JOINS TO WHAT ------------------- -Edges come from digests: an input whose sha256 equals some task's output -sha256 was produced by that task. An input matching no output came from -outside the run. - -Declared edges from definition.json fill the gaps, because two kinds of -artifact have no digest. Folders, which the engine cannot hash, and -subworkflow ports, which are boundary placeholders with no file on disk. -Without the declared fallback those tasks would read as edgeless. - -Digests alone are not enough for a different reason too: a task that copies -its input to its output produces identical bytes, so a pure digest join -reports it as its own producer. Self-edges are dropped. - -HONEST LIMITS -------------- -Skipped tasks are recorded in full, with digests, so a cached run gives the -same graph as a fresh one. That is the point of the format, but it means a -task's own status is often `skipped` rather than `completed`, kept as -`engine_status` and mapped to CACHED, and the record describes outputs -that already existed rather than work just done. - -A record whose `incomplete` names `digests_disabled` or `digests_partial` -has edges this adapter cannot see. Those tasks are still emitted, so they -appear as nodes, and their missing edges fail open to EXTERNAL rather than -being silently dropped. +horus-lineage run directories, read into the graph. + +Horus runs tasks on different machines, so paths cannot join a run back +together. horus-lineage records a sha256 for every input and output on the +machine holding the bytes, and edges join on those: an input matching a +task's output was produced by it, and one matching none is external. +Declared edges from definition.json fill in for folders and subworkflow +ports, which have no digest. A task that copies input to output would match +itself, so self-edges are dropped. + +Layout: run.json, definition.json, one ..json per task or a +records.jsonl. Skipped tasks are recorded with digests and map to CACHED. A +record marked digests_disabled or digests_partial has edges this cannot see; +they fail open to EXTERNAL. """ -import argparse import json +import sys from pathlib import Path from clew.graph.graph import STATUS_CACHED, relative_to, task_status +from clew.contracts import Extractor RECORD_FORMAT = "horus-lineage/v1" PLAN = "run.json" @@ -119,6 +83,18 @@ def script_of(record): return code[0]["path"] if code else "" +def duration_of(task): + """Seconds between started_at and finished_at, when the record has both.""" + from datetime import datetime + try: + start = datetime.fromisoformat(task["started_at"]) + end = datetime.fromisoformat(task["finished_at"]) + except (KeyError, TypeError, ValueError): + return None + seconds = (end - start).total_seconds() + return round(seconds, 3) if seconds >= 0 else None + + def labels_of(entry): """ An artifact's labels, keeping only string keys and values. @@ -222,6 +198,9 @@ def extract(run_dir): workpath = relative_to(record.get("working_dir"), plan.get("run_directory")) if workpath: tasks[node]["workpath"] = workpath + duration = duration_of(task) + if duration is not None: + tasks[node]["metrics"] = {"duration_s": duration} for entry in record.get("inputs", []): digest = entry.get("sha256") @@ -261,31 +240,51 @@ def extract(run_dir): "output_details": output_details} -def main(argv=None): - parser = argparse.ArgumentParser( - description="Build a Clew graph from a horus-lineage run directory.") - parser.add_argument("--run-dir", required=True, - help="a ~/.horus-lineage// directory") - parser.add_argument("--json-out", help="path to write the graph as JSON") - args = parser.parse_args(argv) - - graph = extract(args.run_dir) - known = set(graph["tasks"]) - external = [e for e in graph["edges"] if e["producer"] == "EXTERNAL"] - dangling = [e for e in graph["edges"] - if e["producer"] not in known and e["producer"] != "EXTERNAL"] - skipped = [t for t in graph["tasks"].values() if t["status"] == STATUS_CACHED] - - print(f"tasks in run : {len(graph['tasks'])}") - print(f" skipped (cached) : {len(skipped)}") - print(f"input files (edges): {len(graph['edges'])}") - print(f" external inputs : {len(external)}") - print(f" DANGLING : {len(dangling)}") - - if args.json_out: - Path(args.json_out).write_text(json.dumps(graph, indent=2)) - print(f"\nwrote {args.json_out}") +class Horus(Extractor): + name = "horus" + description = "a horus-lineage run directory" + + def add_arguments(self, parser): + parser.add_argument("--run-dir", required=True, + help="a ~/.horus-lineage// directory") + + def records(self, path): + from clew.extract.runs import recorded_timestamp + path = Path(path) + if (path / PLAN).is_file(): # one run: the sidecar lives beside it + return {"root": path.parent, + "runs": [{"name": path.name, "id": path.name, + "timestamp": recorded_timestamp(path / PLAN)}]} + runs = [c for c in path.iterdir() if (c / PLAN).is_file()] if path.is_dir() else [] + if not runs: + return None + return {"root": path, + "runs": [{"name": c.name, "id": c.name, + "timestamp": recorded_timestamp(c / PLAN), + "mtime": c.stat().st_mtime} for c in runs]} + + def load(self, root, run_id): + return extract(Path(root) / run_id) + + def extract(self, args): + return extract(args.run_dir) + + def summarize(self, graph, args): + known = set(graph["tasks"]) + external = [e for e in graph["edges"] if e["producer"] == "EXTERNAL"] + dangling = [e for e in graph["edges"] + if e["producer"] not in known and e["producer"] != "EXTERNAL"] + skipped = [t for t in graph["tasks"].values() if t["status"] == STATUS_CACHED] + print(f"tasks in run : {len(graph['tasks'])}") + print(f" skipped (cached) : {len(skipped)}") + print(f"input files (edges): {len(graph['edges'])}") + print(f" external inputs : {len(external)}") + print(f" DANGLING : {len(dangling)}") + self.coverage(graph) + + +main = Horus.main if __name__ == "__main__": - main() + sys.exit(main()) diff --git a/providers/clew-horus/pyproject.toml b/providers/clew-horus/pyproject.toml new file mode 100644 index 0000000..3042702 --- /dev/null +++ b/providers/clew-horus/pyproject.toml @@ -0,0 +1,24 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "clew-horus" +version = "0.4.0" +description = "Clew provider for horus-lineage run directories" +readme = "README.md" +requires-python = ">=3.9" +license = { text = "AGPL-3.0-only" } +authors = [{ name = "QuietFlare" }] +dependencies = ["clew-lineage>=0.4"] + +[project.entry-points."clew.extractors"] +horus = "clew.provider.horus.extractor_lineage" + +[project.entry-points."clew.adapters"] +vina-docking = "clew.provider.horus.adapter_vina_docking" + +[tool.setuptools.packages.find] +# Installs into the clew.provider namespace; no __init__.py above the provider. +include = ["clew*"] +namespaces = true diff --git a/tests/fixtures/horus_run/analyse.d4185229.json b/providers/clew-horus/tests/fixtures/horus_run/analyse.d4185229.json similarity index 100% rename from tests/fixtures/horus_run/analyse.d4185229.json rename to providers/clew-horus/tests/fixtures/horus_run/analyse.d4185229.json diff --git a/tests/fixtures/horus_run/definition.json b/providers/clew-horus/tests/fixtures/horus_run/definition.json similarity index 100% rename from tests/fixtures/horus_run/definition.json rename to providers/clew-horus/tests/fixtures/horus_run/definition.json diff --git a/tests/fixtures/horus_run/prep.b10b18f9.json b/providers/clew-horus/tests/fixtures/horus_run/prep.b10b18f9.json similarity index 100% rename from tests/fixtures/horus_run/prep.b10b18f9.json rename to providers/clew-horus/tests/fixtures/horus_run/prep.b10b18f9.json diff --git a/tests/fixtures/horus_run/qc.997164d1.json b/providers/clew-horus/tests/fixtures/horus_run/qc.997164d1.json similarity index 100% rename from tests/fixtures/horus_run/qc.997164d1.json rename to providers/clew-horus/tests/fixtures/horus_run/qc.997164d1.json diff --git a/tests/fixtures/horus_run/report.845e9183.json b/providers/clew-horus/tests/fixtures/horus_run/report.845e9183.json similarity index 100% rename from tests/fixtures/horus_run/report.845e9183.json rename to providers/clew-horus/tests/fixtures/horus_run/report.845e9183.json diff --git a/tests/fixtures/horus_run/run.json b/providers/clew-horus/tests/fixtures/horus_run/run.json similarity index 100% rename from tests/fixtures/horus_run/run.json rename to providers/clew-horus/tests/fixtures/horus_run/run.json diff --git a/providers/clew-horus/tests/fixtures/pantheon_vina/definition.json b/providers/clew-horus/tests/fixtures/pantheon_vina/definition.json new file mode 100644 index 0000000..b7f6883 --- /dev/null +++ b/providers/clew-horus/tests/fixtures/pantheon_vina/definition.json @@ -0,0 +1,114 @@ +{ + "name": "AutoDock Vina Docking", + "tasks": [ + { + "id": "prep", + "definition_id": null, + "kind": "horus_task", + "name": "Prepare receptor + ligands + box", + "executor": "conda_python_environment", + "runtime": "python_script", + "target": "local", + "inputs": [ + { + "id": "receptor", + "kind": "file", + "path": "examples/receptor.pdb" + }, + { + "id": "ligands", + "kind": "file", + "path": "examples/ligands.smi" + } + ], + "outputs": [ + { + "id": "vina_inputs", + "kind": "file", + "path": "results/vina_inputs.tar.gz" + } + ] + }, + { + "id": "dock", + "definition_id": null, + "kind": "horus_task", + "name": "Run AutoDock Vina", + "executor": "conda_python_environment", + "runtime": "python_script", + "target": "local", + "inputs": [ + { + "id": "vina_inputs", + "kind": "file", + "path": "results/vina_inputs.tar.gz" + } + ], + "outputs": [ + { + "id": "docking_out", + "kind": "file", + "path": "results/docking_out.tar.gz" + } + ] + }, + { + "id": "summary", + "definition_id": null, + "kind": "horus_task", + "name": "Summarize docking energies", + "executor": "shell", + "runtime": "python_script", + "target": "local", + "inputs": [ + { + "id": "docking_out", + "kind": "file", + "path": "results/docking_out.tar.gz" + } + ], + "outputs": [ + { + "id": "summary", + "kind": "file", + "path": "results/summary.csv" + }, + { + "id": "poses", + "kind": "file", + "path": "results/poses.csv" + } + ] + } + ], + "edges": [ + { + "source": "prep", + "source_output": "vina_inputs", + "target": "dock", + "target_input": "vina_inputs", + "transfer": true + }, + { + "source": "dock", + "source_output": "docking_out", + "target": "summary", + "target_input": "docking_out", + "transfer": true + }, + { + "source": "artifact-ligands", + "source_output": "ligands", + "target": "prep", + "target_input": "ligands", + "transfer": true + }, + { + "source": "artifact-receptor", + "source_output": "receptor", + "target": "prep", + "target_input": "receptor", + "transfer": true + } + ] +} \ No newline at end of file diff --git a/providers/clew-horus/tests/fixtures/pantheon_vina/dock.a864ed01.json b/providers/clew-horus/tests/fixtures/pantheon_vina/dock.a864ed01.json new file mode 100644 index 0000000..4077109 --- /dev/null +++ b/providers/clew-horus/tests/fixtures/pantheon_vina/dock.a864ed01.json @@ -0,0 +1,60 @@ +{ + "format": "horus-lineage/v1", + "run": "01724523d075657e", + "execution": "b9513026e8664a5e8afb6f409a5a2c87", + "definition_sha256": "76a234ed4e2249a8081819824df4ada3b5331e02119790215ed49e12fc68dd9a", + "recorded_at": "2026-09-11T13:01:51.391522+00:00", + "task": { + "id": "dock", + "definition_id": null, + "kind": "horus_task", + "name": "Run AutoDock Vina", + "status": "completed", + "skip_reason": null, + "runs": 1, + "started_at": "2026-09-11T13:01:22.501064+00:00", + "finished_at": "2026-09-11T13:01:51.391499+00:00" + }, + "target": { + "kind": "local", + "location_id": "local://Seemas-MacBook-Pro-2.local" + }, + "working_dir": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/dock/b9513026e8664a5e8afb6f409a5a2c87", + "command": "python /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/dock/b9513026e8664a5e8afb6f409a5a2c87/dock.py --inputs /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/vina_inputs.tar.gz --out /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/docking_out.tar.gz --exhaustiveness 16 --n-poses 9 --cpu 0", + "environment": { + "executor": { + "kind": "conda_python_environment", + "sha256": "c3fcdb40c5efc7965742a30edff645ab5e3ee5df039fddf3aca5bc1bb827ddae" + }, + "runtime": { + "kind": "python_script", + "sha256": "e1512c9377223702ee3d3e49a1fef088a0874b6796616e4a539a94802d40ba4a" + }, + "config_sha256": "d3a4e1bbc903b31173a1b243b24205846774972ca3b1a688bd1b3178286fb983" + }, + "code": [ + { + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/scripts/dock.py", + "size": 6303, + "sha256": "3abc82b62943c4904c3f5ca20635b7b9e29baeb8be3235e3ec0a527116050610", + "role": "script" + } + ], + "inputs": [ + { + "id": "vina_inputs", + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/vina_inputs.tar.gz", + "size": 48454, + "sha256": "a10ba72d2633ea962e00a9696d0efb8863cb2783e08b621efacabcd64b195041" + } + ], + "outputs": [ + { + "id": "docking_out", + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/docking_out.tar.gz", + "size": 6766, + "sha256": "2ef3446492e8c8c0c703836755694d94fe227c982916ecfa0efc8d266bc699e1" + } + ], + "incomplete": [] +} \ No newline at end of file diff --git a/providers/clew-horus/tests/fixtures/pantheon_vina/ligands.smi b/providers/clew-horus/tests/fixtures/pantheon_vina/ligands.smi new file mode 100644 index 0000000..cbd1e25 --- /dev/null +++ b/providers/clew-horus/tests/fixtures/pantheon_vina/ligands.smi @@ -0,0 +1,3 @@ +Cc1ccc(cc1Nc1nccc(n1)-c1cccnc1)NC(=O)c1ccc(cc1)CN1CCN(C)CC1 imatinib +Cn1cnc2c1c(=O)n(C)c(=O)n2C caffeine +CC(=O)Oc1ccccc1C(=O)O aspirin diff --git a/providers/clew-horus/tests/fixtures/pantheon_vina/prep.b10b18f9.json b/providers/clew-horus/tests/fixtures/pantheon_vina/prep.b10b18f9.json new file mode 100644 index 0000000..d5efda9 --- /dev/null +++ b/providers/clew-horus/tests/fixtures/pantheon_vina/prep.b10b18f9.json @@ -0,0 +1,66 @@ +{ + "format": "horus-lineage/v1", + "run": "01724523d075657e", + "execution": "9fa19c5a8f7b41deb940816a425297da", + "definition_sha256": "76a234ed4e2249a8081819824df4ada3b5331e02119790215ed49e12fc68dd9a", + "recorded_at": "2026-09-11T13:01:22.498538+00:00", + "task": { + "id": "prep", + "definition_id": null, + "kind": "horus_task", + "name": "Prepare receptor + ligands + box", + "status": "completed", + "skip_reason": null, + "runs": 1, + "started_at": "2026-09-11T13:00:57.015985+00:00", + "finished_at": "2026-09-11T13:01:22.498515+00:00" + }, + "target": { + "kind": "local", + "location_id": "local://Seemas-MacBook-Pro-2.local" + }, + "working_dir": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/prep/9fa19c5a8f7b41deb940816a425297da", + "command": "python /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/prep/9fa19c5a8f7b41deb940816a425297da/prep.py --receptor /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/examples/receptor.pdb --ligands /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/examples/ligands.smi --out /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/vina_inputs.tar.gz --center 15.190 53.903 16.917 --size 20 20 20", + "environment": { + "executor": { + "kind": "conda_python_environment", + "sha256": "5584981f3202cabc7158c025bf22799f2ed530fd04250c1a4106c2128fe745e9" + }, + "runtime": { + "kind": "python_script", + "sha256": "a747af7633414f543517b65e651770789c89daa688e072b8492c12b658a1742d" + }, + "config_sha256": "237aa866a63317497ceb25e8c3e2b85c0c0cb7cd39e638933ae19fccd1699573" + }, + "code": [ + { + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/scripts/prep.py", + "size": 13267, + "sha256": "00433c43fceed003d96a90d7f8e76d31ca2b9992b341a519647782fbbb5e0d2a", + "role": "script" + } + ], + "inputs": [ + { + "id": "receptor", + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/examples/receptor.pdb", + "size": 180634, + "sha256": "7935d403d08485691246810f0eebe36bc68981d60fee77021a5f365e5c68170b" + }, + { + "id": "ligands", + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/examples/ligands.smi", + "size": 135, + "sha256": "4602ae4a1ccfcb8e129941473f62db2af5ee1461955221e2f360594355126818" + } + ], + "outputs": [ + { + "id": "vina_inputs", + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/vina_inputs.tar.gz", + "size": 48454, + "sha256": "a10ba72d2633ea962e00a9696d0efb8863cb2783e08b621efacabcd64b195041" + } + ], + "incomplete": [] +} \ No newline at end of file diff --git a/providers/clew-horus/tests/fixtures/pantheon_vina/run.json b/providers/clew-horus/tests/fixtures/pantheon_vina/run.json new file mode 100644 index 0000000..a342fe4 --- /dev/null +++ b/providers/clew-horus/tests/fixtures/pantheon_vina/run.json @@ -0,0 +1,47 @@ +{ + "format": "horus-lineage/v1", + "run": "01724523d075657e", + "run_scope": null, + "run_directory": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results", + "workflow": { + "id": "fd048832-c03b-4930-84bd-0c65e14c2710", + "slug": null, + "name": "AutoDock Vina Docking" + }, + "started_at": "2026-09-11T13:00:57.015337+00:00", + "finished_at": "2026-09-11T13:01:51.486135+00:00", + "status": "completed", + "definition": { + "file": "definition.json", + "sha256": "76a234ed4e2249a8081819824df4ada3b5331e02119790215ed49e12fc68dd9a" + }, + "source": { + "file": "workflow.yaml", + "sha256": "9be209d58ffa9e71bccb2b90743ff84b90251570a205ce1514735a1f64866e85" + }, + "code": [ + { + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/scripts/dock.py", + "size": 6303, + "sha256": "3abc82b62943c4904c3f5ca20635b7b9e29baeb8be3235e3ec0a527116050610", + "role": "script" + }, + { + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/scripts/prep.py", + "size": 13267, + "sha256": "00433c43fceed003d96a90d7f8e76d31ca2b9992b341a519647782fbbb5e0d2a", + "role": "script" + }, + { + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/scripts/summary.py", + "size": 6654, + "sha256": "c390f67b6ecbdd504650bcc691d7f78bd93ba3bb396678a56c9c21220cf8a757", + "role": "script" + } + ], + "tasks": [ + "prep", + "dock", + "summary" + ] +} \ No newline at end of file diff --git a/providers/clew-horus/tests/fixtures/pantheon_vina/summary.761b7ad8.json b/providers/clew-horus/tests/fixtures/pantheon_vina/summary.761b7ad8.json new file mode 100644 index 0000000..a04bcef --- /dev/null +++ b/providers/clew-horus/tests/fixtures/pantheon_vina/summary.761b7ad8.json @@ -0,0 +1,66 @@ +{ + "format": "horus-lineage/v1", + "run": "01724523d075657e", + "execution": "2848c1504e5142ef873cdf5fa3829176", + "definition_sha256": "76a234ed4e2249a8081819824df4ada3b5331e02119790215ed49e12fc68dd9a", + "recorded_at": "2026-09-11T13:01:51.484775+00:00", + "task": { + "id": "summary", + "definition_id": null, + "kind": "horus_task", + "name": "Summarize docking energies", + "status": "completed", + "skip_reason": null, + "runs": 1, + "started_at": "2026-09-11T13:01:51.392738+00:00", + "finished_at": "2026-09-11T13:01:51.484752+00:00" + }, + "target": { + "kind": "local", + "location_id": "local://Seemas-MacBook-Pro-2.local" + }, + "working_dir": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/summary/2848c1504e5142ef873cdf5fa3829176", + "command": "python3 /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/summary/2848c1504e5142ef873cdf5fa3829176/summary.py --docking /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/docking_out.tar.gz --summary /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/summary.csv --poses /private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/poses.csv", + "environment": { + "executor": { + "kind": "shell", + "sha256": "6fa3b9ca39d6f52c325cd7668bea883e571e3092b9261f03820c8dbbfe49f9c3" + }, + "runtime": { + "kind": "python_script", + "sha256": "21a89a94e8fab7ec0798a7beda9da071fdde2706c16e4055cb57fec1e28d80a5" + }, + "config_sha256": "bfca87207b58876eeb4ae9aba325ce9396de76c71431e7e44ef738bd99d97373" + }, + "code": [ + { + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/scripts/summary.py", + "size": 6654, + "sha256": "c390f67b6ecbdd504650bcc691d7f78bd93ba3bb396678a56c9c21220cf8a757", + "role": "script" + } + ], + "inputs": [ + { + "id": "docking_out", + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/docking_out.tar.gz", + "size": 6766, + "sha256": "2ef3446492e8c8c0c703836755694d94fe227c982916ecfa0efc8d266bc699e1" + } + ], + "outputs": [ + { + "id": "summary", + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/summary.csv", + "size": 148, + "sha256": "b6ca0d814482974a49525c30f2984e177d016443eb1935d5692cb6f43057f485" + }, + { + "id": "poses", + "path": "/private/tmp/claude-501/-Users-seemajagadeesh-workspace-BioTechJourney-clew/4edf8cb5-4d50-46bc-88ef-972a18882a45/scratchpad/pantheon/workflows/drug-discovery/w02-autodock-vina-docking/horus_workflow_results/results/poses.csv", + "size": 677, + "sha256": "31715bb7e94709d4b7e4d565c0b1bb0bfa58211c00995bd9e9a671c20fe3f871" + } + ], + "incomplete": [] +} \ No newline at end of file diff --git a/providers/clew-horus/tests/fixtures/pantheon_vina/workflow.yaml b/providers/clew-horus/tests/fixtures/pantheon_vina/workflow.yaml new file mode 100644 index 0000000..202c284 --- /dev/null +++ b/providers/clew-horus/tests/fixtures/pantheon_vina/workflow.yaml @@ -0,0 +1,150 @@ +kind: horus_workflow +name: AutoDock Vina Docking +artifacts: + - id: ligands + name: Candidate ligand library + description: SMILES (or SDF) file listing candidate ligands to dock against the receptor. + kind: file + path: examples/ligands.smi + + - id: receptor + name: Target receptor structure + description: PDB file of the target receptor protein to dock ligands into. + kind: file + path: examples/receptor.pdb + +tasks: + - kind: horus_task + id: prep + name: Prepare receptor + ligands + box + description: Converts the receptor and ligands to PDBQT format and defines the docking search box for AutoDock Vina. + skip_if_complete: false + inputs: + - id: receptor + name: Target receptor structure + kind: file + path: examples/receptor.pdb + - id: ligands + name: Candidate ligand library + kind: file + path: examples/ligands.smi + outputs: + - id: vina_inputs + name: Vina input archive + path: results/vina_inputs.tar.gz + kind: file + + resources: + cpus: 2 + memory_gb: 4 + executor: + kind: conda_python_environment + environment_dir: .horus_python_environment + conda: micromamba # set to mamba/micromamba if that's what's on PATH + python_version: "3.11" + channels: + - conda-forge + conda_requirements: + - openbabel # receptor -> rigid PDBQT (obabel) + - meeko # ligand -> PDBQT (mk_prepare_ligand.py) + - rdkit # 3D embedding of SMILES + Meeko's typing + - numpy + - scipy + - gemmi + runtime: + kind: python_script + script: scripts/prep.py + # Docking box: edit --center / --size for your pocket. Drop --center to + # derive it from a --ref-ligand, or omit both for blind (whole-receptor) + # docking (then use a large --size). Default here is the 1iep ATP pocket. + args: >- + --receptor ${receptor} --ligands ${ligands} --out ${vina_inputs} --center 15.190 53.903 16.917 --size 20 20 20 + target: + kind: local + - kind: horus_task + id: dock + name: Run AutoDock Vina + description: Docks each ligand into the defined search box using AutoDock Vina and records the resulting binding poses. + inputs: + - id: vina_inputs + name: Vina input archive + path: results/vina_inputs.tar.gz + kind: file + outputs: + - id: docking_out + name: Docking results archive + path: results/docking_out.tar.gz + kind: file + + # Advisory only: read by resource-aware targets (Slurm, Terraform) and by + # the tc-os dashboard to compare what was asked for against what was used. + resources: + cpus: 16 + memory_gb: 32 + executor: + kind: conda_python_environment + environment_dir: .horus_python_environment + conda: micromamba # set to mamba/micromamba if that's what's on PATH + python_version: "3.11" + channels: + - conda-forge + conda_requirements: + - vina # AutoDock Vina + Python bindings (conda-forge) + - numpy + runtime: + kind: python_script + script: scripts/dock.py + args: >- + --inputs ${vina_inputs} --out ${docking_out} --exhaustiveness 16 --n-poses 9 --cpu 0 + target: + kind: local + - kind: horus_task + id: summary + name: Summarize docking energies + description: Parses the Vina docking output into ranked per-ligand and per-pose binding energy tables. + inputs: + - id: docking_out + name: Docking results archive + path: results/docking_out.tar.gz + kind: file + outputs: + - id: summary + name: Ranked energy summary + path: results/summary.csv + kind: file + - id: poses + name: Per-pose energy table + path: results/poses.csv + kind: file + executor: + kind: shell + runtime: + kind: python_script + script: scripts/summary.py + python: python3 # summary is stdlib-only; use the target's python3 + args: --docking ${docking_out} --summary ${summary} --poses ${poses} + target: + kind: local +edges: + - source: prep + source_output: vina_inputs + target: dock + target_input: vina_inputs + - source: dock + source_output: docking_out + target: summary + target_input: docking_out + + - source: artifact-ligands + source_output: ligands + target: prep + target_input: ligands + + - source: artifact-receptor + source_output: receptor + target: prep + target_input: receptor + +orchestrator_target: + kind: local + working_directory: horus_workflow_results diff --git a/tests/test_horus.py b/providers/clew-horus/tests/test_horus.py similarity index 98% rename from tests/test_horus.py rename to providers/clew-horus/tests/test_horus.py index d4ddaaa..36b77e4 100644 --- a/tests/test_horus.py +++ b/providers/clew-horus/tests/test_horus.py @@ -19,9 +19,9 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.extract import horus as hz +from clew.provider.horus import extractor_lineage as hz from clew.graph import blast_radius as core FIXTURES = Path(__file__).resolve().parent / "fixtures" diff --git a/providers/clew-horus/tests/test_horus_contract.py b/providers/clew-horus/tests/test_horus_contract.py new file mode 100644 index 0000000..674aaaf --- /dev/null +++ b/providers/clew-horus/tests/test_horus_contract.py @@ -0,0 +1,29 @@ +"""What this extractor emits is the one graph every question reads.""" + +import unittest +from pathlib import Path + +from clew.graph.graph import STATUSES, contract_violations +from clew.provider.horus import extractor_lineage as hz + +FIXTURES = Path(__file__).resolve().parent / "fixtures" + + +class Contract(unittest.TestCase): + def graphs(self): + yield "horus", hz.extract(FIXTURES / "horus_run") + + def test_conforms(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + self.assertEqual(contract_violations(graph), []) + + def test_statuses_are_in_the_vocabulary(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + for task in graph["tasks"].values(): + self.assertIn(task["status"], STATUSES) + + +if __name__ == "__main__": + unittest.main() diff --git a/providers/clew-horus/tests/test_pantheon_vina.py b/providers/clew-horus/tests/test_pantheon_vina.py new file mode 100644 index 0000000..ba5716b --- /dev/null +++ b/providers/clew-horus/tests/test_pantheon_vina.py @@ -0,0 +1,98 @@ +""" +Pantheon W-02, AutoDock Vina docking, run under Horus with horus-lineage on +2026-09-11: three ligands, three tasks. The fixture is that run's record. +""" + +import argparse +import io +import json +import unittest +from contextlib import redirect_stdout, redirect_stderr +from pathlib import Path + +from clew.graph import blast_radius as core +from clew.provider.horus import extractor_lineage as hz +from clew.provider.horus.adapter_vina_docking import VinaDocking +from clew.questions import impact + +FIXTURE = Path(__file__).resolve().parent / "fixtures" / "pantheon_vina" +LIGANDS = str(FIXTURE / "ligands.smi") + + +def run_impact(*argv): + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + impact.main(["--pipeline", "vina-docking", "--graph", str(FIXTURE / "graph.json"), *argv]) + text = out.getvalue() + return {i["task"].split("/")[1]: i for i in json.loads(text[text.index("{"):])["plan"]} + + +class TestVinaRun(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.graph = hz.extract(FIXTURE) + (FIXTURE / "graph.json").write_text(json.dumps(cls.graph)) + cls.adapter = VinaDocking() + cls.kind = cls.adapter.triggers["ligand"] + + @classmethod + def tearDownClass(cls): + (FIXTURE / "graph.json").unlink(missing_ok=True) + + def test_three_tasks_and_the_library_enters_at_prep(self): + by_process = {t["process"]: h for h, t in self.graph["tasks"].items()} + self.assertEqual(sorted(by_process), ["dock", "prep", "summary"]) + entry = self.kind.resolve(self.graph, "caffeine", argparse.Namespace(ligands=LIGANDS)) + self.assertEqual(sorted(entry), ["aspirin", "caffeine", "imatinib"]) + self.assertEqual(entry["caffeine"], [by_process["prep"]]) + + def test_every_ligand_reaches_everything_and_owns_nothing(self): + entry = self.kind.resolve(self.graph, None, argparse.Namespace(ligands=LIGANDS)) + radius = core.blast_radius(self.graph, entry) + for ligand, r in radius.items(): + self.assertEqual(len(r["affected"]), 3, ligand) + self.assertEqual(r["exclusive"], set(), ligand) + + def test_withdrawing_a_ligand_is_separable_everywhere(self): + items = run_impact("--trigger", "ligand:caffeine", "--ligands", LIGANDS, "--json", "-") + for process, item in items.items(): + self.assertEqual(item["contribution"], "SEPARABLE", process) + self.assertEqual(item["class_asserted_by"], "vina-docking", process) + + def test_a_defect_in_a_step_is_not_per_ligand(self): + items = run_impact("--trigger", "process:dock", "--json", "-") + self.assertEqual(items["dock"]["contribution"], "REGENERABLE") + self.assertNotIn("class_asserted_by", items["dock"]) + + def test_recorded_run_time_reaches_the_plan(self): + by_process = {t["process"]: t for t in self.graph["tasks"].values()} + self.assertAlmostEqual(by_process["dock"]["metrics"]["duration_s"], 28.89, places=1) + items = run_impact("--trigger", "process:dock", "--json", "-") + self.assertAlmostEqual(items["dock"]["metrics"]["duration_s"], 28.89, places=1) + + def test_a_step_defect_costs_the_recorded_time_of_what_is_rerun(self): + # Storage must be verified for a verdict to settle to REGENERATE, so + # stand in for the run's work tree with the recorded workpaths. + import tempfile + with tempfile.TemporaryDirectory() as work: + for task in self.graph["tasks"].values(): + (Path(work) / task["workpath"]).mkdir(parents=True) + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + impact.main(["--pipeline", "vina-docking", "--graph", str(FIXTURE / "graph.json"), + "--trigger", "process:dock", "--work-root", work, "--json", "-"]) + text = out.getvalue() + cost = json.loads(text[text.index("{"):])["cost"] + bucket = cost["by_action"]["REGENERATE"] + self.assertEqual(bucket["tasks"], 2) # dock and summary + self.assertAlmostEqual(bucket["metrics"]["duration_s"], 28.96, places=1) + self.assertEqual(bucket["missing"], {"duration_s": 0}) + self.assertIn("COST OF THIS PLAN", text) + + def test_the_gate_lists_the_ligands(self): + self.assertEqual(self.kind.values(argparse.Namespace(ligands=LIGANDS)), + ["aspirin", "caffeine", "imatinib"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/providers/clew-horus/tests/test_runs_horus.py b/providers/clew-horus/tests/test_runs_horus.py new file mode 100644 index 0000000..1c6e1af --- /dev/null +++ b/providers/clew-horus/tests/test_runs_horus.py @@ -0,0 +1,37 @@ +"""--runs on a horus-lineage record: one run directory, or a root of them.""" + +import shutil +import tempfile +import unittest +from pathlib import Path + +from clew.extract import runs + +FIXTURES = Path(__file__).resolve().parent / "fixtures" + + +class HorusRuns(unittest.TestCase): + def test_a_single_run_directory(self): + store = runs.Runs(FIXTURES / "horus_run") + self.assertEqual(store.kind, "horus") + self.assertEqual(store.records()[0]["timestamp"], "2026-09-02T09:01:34.050766+00:00") + g = store.load() + self.assertEqual(g["run"]["name"], "horus_run") + self.assertTrue(any(d.get("digest", "").startswith("sha256:") + for ds in g["output_details"].values() for d in ds)) + + def test_a_root_of_run_directories(self): + root = Path(tempfile.mkdtemp()) + try: + shutil.copytree(FIXTURES / "horus_run", root / "run-1") + shutil.copytree(FIXTURES / "horus_run", root / "run-2") + store = runs.Runs(root) + self.assertEqual(store.kind, "horus") + self.assertEqual(sorted(n for n, _, _ in store.names()), ["run-1", "run-2"]) + self.assertEqual(store.sidecar_path("run-1"), root / ".clew" / "run-1.digests.json") + finally: + shutil.rmtree(root) + + +if __name__ == "__main__": + unittest.main() diff --git a/providers/clew-latch/README.md b/providers/clew-latch/README.md new file mode 100644 index 0000000..b5701da --- /dev/null +++ b/providers/clew-latch/README.md @@ -0,0 +1,11 @@ +# clew-latch + +Clew provider for Latch executions, from saved records or the API. + +```bash +pip install clew-latch +``` + +Registers under the entry-point groups Clew discovers. `clew providers` +lists it. Built exactly as a third-party provider would be: see +[providers](../../docs/providers.md) in the engine's docs. diff --git a/providers/clew-latch/clew/provider/latch/__init__.py b/providers/clew-latch/clew/provider/latch/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/clew/extract/latch.py b/providers/clew-latch/clew/provider/latch/extractor_execution.py similarity index 73% rename from clew/extract/latch.py rename to providers/clew-latch/clew/provider/latch/extractor_execution.py index 32da339..f59cf58 100644 --- a/clew/extract/latch.py +++ b/providers/clew-latch/clew/provider/latch/extractor_execution.py @@ -1,27 +1,18 @@ """ -Clew: lineage adapter for Latch executions. - -Latch keeps per-task records behind the GraphQL API its SDK uses: one -execution graph node per task with status, timings, cost and the inputs -and outputs it ran with, and the workflow's commit and image hashes. The -inputs and outputs are Flyte literal maps that name files by latch:// -path, so edges join on that path and two executions stitch at a shared -path the same way two Nextflow runs do. - -Two ways in. `--records DIR` reads a saved execution.json plus one -literals file per node, which is how the fixtures and tests work. -`--execution ID` fetches the same over the API with the token that -`latch login` stores. The API path has not yet been run against a live -execution. The schema comes from introspection; the literal-map shape -comes from Flyte's JSON encoding. - -The task hash is the execution graph node id. The container is -"@", since every task in an execution runs -the workflow's image. The script field carries the workflow commit hash. -Optional `price` and `duration_s` per task. +Latch executions, read into the graph. + +Latch keeps one execution graph node per task behind the GraphQL API its SDK +uses, with status, timings, cost, the Flyte literal maps of inputs and +outputs, and the workflow's commit and image hashes. Files are named by +latch:// path, so edges join on that path. --records DIR reads a saved +execution.json plus one literals file per node. --execution ID fetches the +same with the token latch login stores, and has not yet met a live +execution. + +The task hash is the node id, the container "@", the +script the workflow commit, with optional price and duration_s. """ -import argparse import base64 import json import os @@ -31,6 +22,7 @@ from pathlib import Path from clew.graph.graph import EXTERNAL, task_status +from clew.contracts import Extractor API = "https://vacuole.latch.bio/graphql" TOKEN_FILE = Path.home() / ".latch" / "token" @@ -127,10 +119,10 @@ def extract(records): } price = node.get("cost") if node.get("cost") is not None else node.get("price") if isinstance(price, (int, float)): - task["price"] = price + task.setdefault("metrics", {})["price"] = price duration = seconds_between(node.get("startTime"), node.get("resolutionTime")) if duration is not None: - task["duration_s"] = duration + task.setdefault("metrics", {})["duration_s"] = duration tasks[node_id] = task produced = list(dict.fromkeys( @@ -205,39 +197,37 @@ def fetch(execution_id, token, api=API): return {"execution": execution, "literals": literals} -def main(argv=None): - parser = argparse.ArgumentParser( - description="Build a Clew graph from a Latch execution.") - source = parser.add_mutually_exclusive_group(required=True) - source.add_argument("--records", help="directory with execution.json and literals") - source.add_argument("--execution", help="execution id to fetch over the API") - parser.add_argument("--token", help="API token; default ~/.latch/token") - parser.add_argument("--json-out", help="path to write the graph as JSON") - args = parser.parse_args(argv) +class Latch(Extractor): + name = "latch" + description = "a Latch execution, from saved records or the API" - if args.records: - records = load_records(args.records) - else: + def add_arguments(self, parser): + source = parser.add_mutually_exclusive_group(required=True) + source.add_argument("--records", help="directory with execution.json and literals") + source.add_argument("--execution", help="execution id to fetch over the API") + parser.add_argument("--token", help="API token; default ~/.latch/token") + + def extract(self, args): + if args.records: + return extract(load_records(args.records)) token = args.token or token_from_env() if not token: print("clew: no API token; pass --token or run latch login", file=sys.stderr) - return 2 - records = fetch(args.execution, token) - - graph = extract(records) - external = [e for e in graph["edges"] if e["producer"] == EXTERNAL] - priced = [t for t in graph["tasks"].values() if "price" in t] - - print(f"tasks : {len(graph['tasks'])}") - print(f"input files (edges): {len(graph['edges'])}") - print(f" external inputs : {len(external)}") - if priced: - print(f"total price : {sum(t['price'] for t in priced):.2f}") - - if args.json_out: - Path(args.json_out).write_text(json.dumps(graph, indent=2)) - print(f"\nwrote {args.json_out}") - return 0 + raise SystemExit(2) + return extract(fetch(args.execution, token)) + + def summarize(self, graph, args): + external = [e for e in graph["edges"] if e["producer"] == EXTERNAL] + priced = [t for t in graph["tasks"].values() if "price" in t.get("metrics", {})] + print(f"tasks : {len(graph['tasks'])}") + print(f"input files (edges): {len(graph['edges'])}") + print(f" external inputs : {len(external)}") + if priced: + print(f"total price : {sum(t['metrics']['price'] for t in priced):.2f}") + self.coverage(graph) + + +main = Latch.main if __name__ == "__main__": diff --git a/providers/clew-latch/pyproject.toml b/providers/clew-latch/pyproject.toml new file mode 100644 index 0000000..a1cdd8c --- /dev/null +++ b/providers/clew-latch/pyproject.toml @@ -0,0 +1,21 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "clew-latch" +version = "0.4.0" +description = "Clew provider for Latch executions, from saved records or the API" +readme = "README.md" +requires-python = ">=3.9" +license = { text = "AGPL-3.0-only" } +authors = [{ name = "QuietFlare" }] +dependencies = ["clew-lineage>=0.4"] + +[project.entry-points."clew.extractors"] +latch = "clew.provider.latch.extractor_execution" + +[tool.setuptools.packages.find] +# Installs into the clew.provider namespace; no __init__.py above the provider. +include = ["clew*"] +namespaces = true diff --git a/tests/fixtures/latch/1001.inputs.json b/providers/clew-latch/tests/fixtures/latch/1001.inputs.json similarity index 100% rename from tests/fixtures/latch/1001.inputs.json rename to providers/clew-latch/tests/fixtures/latch/1001.inputs.json diff --git a/tests/fixtures/latch/1001.outputs.json b/providers/clew-latch/tests/fixtures/latch/1001.outputs.json similarity index 100% rename from tests/fixtures/latch/1001.outputs.json rename to providers/clew-latch/tests/fixtures/latch/1001.outputs.json diff --git a/tests/fixtures/latch/1002.inputs.json b/providers/clew-latch/tests/fixtures/latch/1002.inputs.json similarity index 100% rename from tests/fixtures/latch/1002.inputs.json rename to providers/clew-latch/tests/fixtures/latch/1002.inputs.json diff --git a/tests/fixtures/latch/1002.outputs.json b/providers/clew-latch/tests/fixtures/latch/1002.outputs.json similarity index 100% rename from tests/fixtures/latch/1002.outputs.json rename to providers/clew-latch/tests/fixtures/latch/1002.outputs.json diff --git a/tests/fixtures/latch/1003.inputs.json b/providers/clew-latch/tests/fixtures/latch/1003.inputs.json similarity index 100% rename from tests/fixtures/latch/1003.inputs.json rename to providers/clew-latch/tests/fixtures/latch/1003.inputs.json diff --git a/tests/fixtures/latch/1003.outputs.json b/providers/clew-latch/tests/fixtures/latch/1003.outputs.json similarity index 100% rename from tests/fixtures/latch/1003.outputs.json rename to providers/clew-latch/tests/fixtures/latch/1003.outputs.json diff --git a/tests/fixtures/latch/1004.inputs.json b/providers/clew-latch/tests/fixtures/latch/1004.inputs.json similarity index 100% rename from tests/fixtures/latch/1004.inputs.json rename to providers/clew-latch/tests/fixtures/latch/1004.inputs.json diff --git a/tests/fixtures/latch/1004.outputs.json b/providers/clew-latch/tests/fixtures/latch/1004.outputs.json similarity index 100% rename from tests/fixtures/latch/1004.outputs.json rename to providers/clew-latch/tests/fixtures/latch/1004.outputs.json diff --git a/tests/fixtures/latch/execution.json b/providers/clew-latch/tests/fixtures/latch/execution.json similarity index 100% rename from tests/fixtures/latch/execution.json rename to providers/clew-latch/tests/fixtures/latch/execution.json diff --git a/tests/test_latch.py b/providers/clew-latch/tests/test_latch.py similarity index 89% rename from tests/test_latch.py rename to providers/clew-latch/tests/test_latch.py index 24a0e5c..e5ea9e1 100644 --- a/tests/test_latch.py +++ b/providers/clew-latch/tests/test_latch.py @@ -12,9 +12,9 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.extract import latch as lt +from clew.provider.latch import extractor_execution as lt from clew.graph import blast_radius as core from clew.graph.graph import contract_violations, external_input_entry_nodes @@ -77,13 +77,13 @@ def test_names_process_container_and_script(self): self.assertEqual(task["engine_status"], "SUCCEEDED") def test_price_prefers_cost_then_price(self): - self.assertEqual(self.graph["tasks"][ALIGN]["price"], 0.12) - self.assertEqual(self.graph["tasks"][CALL]["price"], 0.30) - self.assertNotIn("price", self.graph["tasks"][REPORT]) + self.assertEqual(self.graph["tasks"][ALIGN]["metrics"]["price"], 0.12) + self.assertEqual(self.graph["tasks"][CALL]["metrics"]["price"], 0.30) + self.assertNotIn("price", self.graph["tasks"][REPORT].get("metrics", {})) def test_duration_from_timestamps(self): - self.assertEqual(self.graph["tasks"][CALL]["duration_s"], 1800) - self.assertEqual(self.graph["tasks"][REPORT]["duration_s"], 60) + self.assertEqual(self.graph["tasks"][CALL]["metrics"]["duration_s"], 1800) + self.assertEqual(self.graph["tasks"][REPORT]["metrics"]["duration_s"], 60) class Triggers(unittest.TestCase): @@ -156,7 +156,9 @@ def test_execution_without_token_fails_cleanly(self): lt.TOKEN_FILE = Path("/nonexistent/token") try: with contextlib.redirect_stderr(io.StringIO()): - self.assertEqual(lt.main(["--execution", "1"]), 2) + with self.assertRaises(SystemExit) as stop: + lt.main(["--execution", "1"]) + self.assertEqual(stop.exception.code, 2) finally: lt.TOKEN_FILE = original if saved is not None: diff --git a/providers/clew-latch/tests/test_latch_contract.py b/providers/clew-latch/tests/test_latch_contract.py new file mode 100644 index 0000000..a7802eb --- /dev/null +++ b/providers/clew-latch/tests/test_latch_contract.py @@ -0,0 +1,29 @@ +"""What this extractor emits is the one graph every question reads.""" + +import unittest +from pathlib import Path + +from clew.graph.graph import STATUSES, contract_violations +from clew.provider.latch import extractor_execution as lt + +FIXTURES = Path(__file__).resolve().parent / "fixtures" / "latch" + + +class Contract(unittest.TestCase): + def graphs(self): + yield "latch", lt.extract(lt.load_records(FIXTURES)) + + def test_conforms(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + self.assertEqual(contract_violations(graph), []) + + def test_statuses_are_in_the_vocabulary(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + for task in graph["tasks"].values(): + self.assertIn(task["status"], STATUSES) + + +if __name__ == "__main__": + unittest.main() diff --git a/providers/clew-nextflow/README.md b/providers/clew-nextflow/README.md new file mode 100644 index 0000000..f356b6c --- /dev/null +++ b/providers/clew-nextflow/README.md @@ -0,0 +1,11 @@ +# clew-nextflow + +Clew provider for Nextflow: the .lineage store, work/ symlinks, Workflow Run RO-Crates, and the nf-core pipeline adapters. + +```bash +pip install clew-nextflow +``` + +Registers under the entry-point groups Clew discovers. `clew providers` +lists it. Built exactly as a third-party provider would be: see +[providers](../../docs/providers.md) in the engine's docs. diff --git a/providers/clew-nextflow/clew/provider/nextflow/__init__.py b/providers/clew-nextflow/clew/provider/nextflow/__init__.py new file mode 100644 index 0000000..953556a --- /dev/null +++ b/providers/clew-nextflow/clew/provider/nextflow/__init__.py @@ -0,0 +1,5 @@ +"""from clew.provider.nextflow import NextflowAdapter""" + +from .adapter import NextflowAdapter + +__all__ = ["NextflowAdapter"] diff --git a/providers/clew-nextflow/clew/provider/nextflow/adapter.py b/providers/clew-nextflow/clew/provider/nextflow/adapter.py new file mode 100644 index 0000000..862967e --- /dev/null +++ b/providers/clew-nextflow/clew/provider/nextflow/adapter.py @@ -0,0 +1,119 @@ +""" +Nextflow names each task "PROCESS (tag)" and nf-core launches from a CSV +samplesheet whose ids appear in those tags. SheetKind joins the two: + + triggers = {"patient": SheetKind(column="patient", members="sample")} +""" + +import csv +import re + +from clew.contracts import REMOVE, Adapter, Trigger + +# The trailing parenthetical on a task's display name: "ALIGN (ID-003)". +# Many tasks use the same shape for other things, so a captured value is +# only accepted when it matches an id from the sheet. +TAG_PATTERN = re.compile(r"\(([^()]+)\)\s*$") + + +def read_sheet(path, column, members=None): + """{id: [member ids]} from one column. One id under two owners is refused.""" + ids = {} + with open(path, newline="") as handle: + for row in csv.DictReader(handle): + owner = (row.get(column) or "").strip() + if not owner: + continue + ids.setdefault(owner, []) + if members: + member = (row.get(members) or "").strip() + if member and member not in ids[owner]: + ids[owner].append(member) + owners = {} + shared = {} + for owner, names in ids.items(): + for label in [owner] + names: + if label in owners and owners[label] != owner: + shared.setdefault(label, {owners[label]}).add(owner) + owners.setdefault(label, owner) + if shared: + listing = "; ".join(f"{label!r} under {', '.join(sorted(owners))}" + for label, owners in sorted(shared.items())) + raise SystemExit(f"{path}: the same id appears under more than one " + f"{column}, so tasks tagged with it cannot be attributed: {listing}") + return ids + + +def task_tag(task): + """The trailing parenthetical of a task's display name, or None.""" + match = TAG_PATTERN.search(task.get("name", "")) + return match.group(1).strip() if match else None + + +def owner_of(tag, label_to_owner): + """The id a tag belongs to. Lane suffixes ("ID-003-L1") match at a separator only, longest label first.""" + if tag in label_to_owner: + return label_to_owner[tag] + for label in sorted(label_to_owner, key=len, reverse=True): + for separator in ("-", "_", "."): + if tag.startswith(label + separator): + return label_to_owner[label] + return None + + +def entry_nodes_by_tag(graph, ids): + """{id: [tasks tagged with it or a member]}. Over-include; under-reporting is the failure.""" + label_to_owner = {} + for owner, names in ids.items(): + label_to_owner[owner] = owner + for name in names: + label_to_owner[name] = owner + entry = {owner: set() for owner in ids} + for task_hash, task in graph["tasks"].items(): + tag = task_tag(task) + if not tag: + continue + owner = owner_of(tag, label_to_owner) + if owner: + entry[owner].add(task_hash) + return {owner: sorted(nodes) for owner, nodes in entry.items()} + + +class SheetKind(Trigger): + """Ids from a samplesheet column, found in task tags. A removal.""" + + mode = REMOVE + + def __init__(self, column, members=None, locate=None): + self.column, self.members = column, members + self.locate = locate # optional: graph -> sheet path, when the site knows where runs keep it + + def add_arguments(self, parser): + parser.add_argument("--samplesheet", metavar="CSV", + help=f"the run's samplesheet; ids come from its {self.column!r} column") + + def sheet(self, args, graph=None): + path = getattr(args, "samplesheet", None) or (self.locate(graph) if self.locate and graph else None) + if not path: + raise SystemExit(f"--samplesheet is required: ids of this kind come from its " + f"{self.column!r} column") + return path + + def ids(self, path): + return read_sheet(path, self.column, self.members) + + def entries(self, graph, ids): + return entry_nodes_by_tag(graph, ids) + + def resolve(self, graph, value, args): + entry = self.entries(graph, self.ids(self.sheet(args, graph))) + if value is not None and value not in entry: + raise SystemExit(f"unknown {self.column} {value!r}; known: {', '.join(sorted(entry))}") + return entry + + def values(self, args, graph=None): + return sorted(self.ids(self.sheet(args, graph))) + + +class NextflowAdapter(Adapter): + """Set `name`, `triggers` and `load_bearing_inputs`.""" diff --git a/providers/clew-nextflow/clew/provider/nextflow/adapter_rnaseq.py b/providers/clew-nextflow/clew/provider/nextflow/adapter_rnaseq.py new file mode 100644 index 0000000..8c65690 --- /dev/null +++ b/providers/clew-nextflow/clew/provider/nextflow/adapter_rnaseq.py @@ -0,0 +1,21 @@ +""" +nf-core/rnaseq. The subject is the sample. The frequent trigger is the +annotation: a GTF bump invalidates every count matrix and differential +result computed against it, while alignments to the unchanged genome +sequence may survive. +""" + +from . import adapter as base + + +class Rnaseq(base.NextflowAdapter): + name = "rnaseq" + triggers = {"sample": base.SheetKind(column="sample")} + # Basenames as recorded in a real 3.26.0 run (iGenomes R64-1-1). The GTF + # has one direct consumer and 149 of 171 tasks in its blast radius. + load_bearing_inputs = ( + "genome.fa", + "genes.gtf", + "genes.bed", + ) + diff --git a/providers/clew-nextflow/clew/provider/nextflow/adapter_sarek.py b/providers/clew-nextflow/clew/provider/nextflow/adapter_sarek.py new file mode 100644 index 0000000..d1035bb --- /dev/null +++ b/providers/clew-nextflow/clew/provider/nextflow/adapter_sarek.py @@ -0,0 +1,24 @@ +""" +nf-core/sarek. The subject is the patient, who can contribute several +samples (normal and tumour). Donor identity comes from the samplesheet, +never from file contents: five donors' alignment tasks all consumed files +named test_1.fastq.gz on a real run. +""" + +from . import adapter as base + + +class Sarek(base.NextflowAdapter): + name = "sarek" + # One patient, several samples (normal, tumour); a tag naming either resolves to the patient. + triggers = {"patient": base.SheetKind(column="patient", members="sample")} + # The reference bundle. Invalidating any of it reaches everything + # calibrated against it. + load_bearing_inputs = ( + "genome.fasta", + "genome.fasta.fai", + "genome.dict", + "dbsnp_146.hg38.vcf.gz", + "mills_and_1000G.indels.vcf.gz", + ) + diff --git a/providers/clew-nextflow/clew/provider/nextflow/adapter_viralrecon.py b/providers/clew-nextflow/clew/provider/nextflow/adapter_viralrecon.py new file mode 100644 index 0000000..38d60af --- /dev/null +++ b/providers/clew-nextflow/clew/provider/nextflow/adapter_viralrecon.py @@ -0,0 +1,24 @@ +""" +nf-core/viralrecon. The subject is the specimen (an ENA run accession such +as ERR10000000); there is no donor concept and none is invented. The +everyday triggers are a contaminated or swapped specimen, and reference +or primer-scheme updates. +""" + +from . import adapter as base + + +class Viralrecon(base.NextflowAdapter): + name = "viralrecon" + triggers = {"sample": base.SheetKind(column="sample")} + # Basenames from a real 2.6.0 run (219-task COG-UK run). Other releases + # name the same files differently, which is why we match what the + # store recorded, not what a config promises. + load_bearing_inputs = ( + "nCoV-2019.reference.fasta", + "nCoV-2019.primer.bed", + "GCA_009858895.3_ASM985889v3_genomic.200409.gff.gz", + "kraken2_human.tar.gz", + "nextclade_sars-cov-2_MN908947_2022-06-14T12_00_00Z.tar.gz", + ) + diff --git a/clew/extract/nextflow_store.py b/providers/clew-nextflow/clew/provider/nextflow/extractor_lineage.py similarity index 63% rename from clew/extract/nextflow_store.py rename to providers/clew-nextflow/clew/provider/nextflow/extractor_lineage.py index c110684..72a85f1 100644 --- a/clew/extract/nextflow_store.py +++ b/providers/clew-nextflow/clew/provider/nextflow/extractor_lineage.py @@ -1,78 +1,27 @@ """ -Clew — lineage adapter for Nextflow's native data lineage store. - -WHY THIS EXISTS ---------------- -The work/ symlink extractor rebuilds history from an accident of staging. -Nextflow's lineage feature records the same facts on purpose: every task, -every output file, and every input reference, written as JSON records into -a .lineage store at run time. When the store exists, it is the better -witness — inputs are typed, externals carry checksums, and each task names -the session it belongs to, so pulling one pipeline out of a shared store is -a field lookup instead of a heuristic. - -Both extractors emit the SAME graph JSON. Everything downstream (clew impact, -core/, domains/) neither knows nor cares which one produced its input. - -THE STORE, AS FOUND ON DISK (lineage/v1beta1) ---------------------------------------------- - .lineage/ - .history/ one line per run: - timestamp \t name \t sessionId \t lid://hash - /.data.json kind: TaskRun - spec: name, sessionId, container, script, - workflowRun, input[] — path inputs are - "lid:///" (internal) - {"path": ..., "checksum": ...} (external) - //.data.json kind: FileOutput - spec.path = absolute path in work/ - #output/ kind: TaskOutput, carries createdAt - #output/ workflow-level outputs; not tasks - -Task hashes here are the full 32 hex characters; work/ folders and the -symlink extractor abbreviate them to "XX/YYYYYY". We abbreviate the same way -so graphs from both extractors are comparable node for node. - -SCOPING: SESSION, NOT RUN -------------------------- -A resumed run reuses cached tasks, and a cached task writes no new record: -the one from the run that first executed it still stands, and still names -that run in `workflowRun`. Filtering by run therefore drops every cached -task, and a fully cached run appears to contain nothing at all. - -`sessionId` is the field that spans a resume chain. Every run in the chain -shares it, so selecting on it returns the tasks the chain actually relied -on. Upstream confirmed this is the intended reading -(nextflow-io/nextflow#7586). - -A chain can hold more than one version of the same task, when a resumed run -invalidated it and ran it again. The version from the later run is live and -the older one is history. Older versions are kept and marked `superseded` -rather than dropped, because their outputs really were produced and may -still be on disk: a deletion plan that omits them under-reports, which is -the one direction that must never happen. Two same-named tasks in one run -are not versions of each other, and a version some task in the session -still reads from is not history either; neither is marked. - -WHAT THIS SOURCE CANNOT SHOW ----------------------------- -The store records no task exit status, so a task that failed is written -exactly like one that succeeded and simply produced no files. Every graph -built here says so in `coverage`, which travels with the graph into -evidence bundles and the dashboard rather than being printed and lost. - -USAGE ------ - clew extract-store --store /path/to/.lineage --list-runs - clew extract-store \ - --store /path/to/.lineage --run tender_mccarthy --json-out graph.json +Nextflow's lineage store, read into the graph. + +Layout (lineage/v1beta1): .history/ lists runs; / +.data.json is a TaskRun whose inputs are lid:/// or an +external path with a checksum; //.data.json is a +FileOutput. Hashes are abbreviated to XX/YYYYYY so graphs from the work tree +and the store compare node for node. + +Scoped by session, not run. A cached task keeps the workflowRun of the run +that first executed it, so filtering by run drops every cached task +(nextflow-io/nextflow#7586). A task a resumed run re-executed is kept and +marked superseded rather than dropped, since its outputs may still be on +disk. + +The store records no exit status; every graph says so in coverage. """ -import argparse import json +import sys from pathlib import Path from clew.graph.graph import STATUS_UNRECORDED +from clew.contracts import Extractor LID_PREFIX = "lid://" @@ -128,7 +77,7 @@ def pick_run(runs, wanted): """ Resolve --run against run name, run-hash prefix, or sessionId prefix. No --run means the most recent run: the common case right after a run - finishes, and the wrong one silently if you meant an older run — which + finishes, and the wrong one silently if you meant an older run, which is why --list-runs exists. """ if not runs: @@ -208,14 +157,10 @@ def consumed_hashes(selected): def superseded_tasks(selected, run_order): """ - Task hashes replaced by a later version of the same task in this chain. - - A version is superseded only by a same-named task from a later run in - `run_order` (run hashes, oldest first). Same run means scatter shards - or a repeated process, not a re-run, and a run the history does not - list cannot be placed, so nothing is claimed for either. A version - that another task in the session still reads from stays live whatever - its age: its outputs are inputs somebody relied on. + Task hashes replaced by a same-named task from a later run in + `run_order`. Same-run repeats are shards, not versions; a run the + history does not list cannot be placed; a version some task still reads + from stays live. None of those is marked. """ position = {run_hash: i for i, run_hash in enumerate(run_order)} consumed = consumed_hashes(selected) @@ -291,12 +236,10 @@ def task_edges(task_hash, spec): def task_outputs(store, task_hash): """ - Collect the task's FileOutput records: ({relative_path: record}, workdir). - - Output records live in subdirectories of the task's store entry, one per - file, nested to mirror the file's path inside the task workdir. The - workdir itself is recovered from any output's absolute path — the store - has no other field for it. + The task's FileOutput records as ({relative_path: record}, workdir). + Records nest under the store entry mirroring the file's path. The + workdir is recovered from an output's absolute path, the store's only + source for it. """ task_dir = Path(store) / task_hash outputs = {} @@ -359,13 +302,10 @@ def coverage_notes(stale, kinds, modes, dangling): def extract(store, session_id): """ - Build the graph for one resume chain, in the exact schema clew - extract-work emits: {"tasks": {...}, "edges": [...], "outputs": {...}} - plus "output_details" (per-output size) and "coverage". - - Scoped by session rather than by run so that cached tasks, which keep - the workflowRun of whichever run first executed them, are still part of - the chain that relied on them. + The graph for one resume chain in the common schema, plus output_details + and coverage. Scoped by session so cached tasks, which keep the + workflowRun that first executed them, stay in the chain that relied on + them. """ versions, kinds, modes = set(), set(), set() selected = [] @@ -434,62 +374,72 @@ def extract(store, session_id): "coverage": coverage_notes(stale, kinds, modes, dangling)} -def main(argv=None): - parser = argparse.ArgumentParser( - description="Build a Clew graph from a Nextflow .lineage store.") - parser.add_argument("--store", required=True, help="path to the .lineage directory") - parser.add_argument("--run", help="run name, run-hash prefix, or sessionId prefix " - "(default: most recent run)") - parser.add_argument("--list-runs", action="store_true", help="list recorded runs and exit") - parser.add_argument("--json-out", help="path to write the graph as JSON") - args = parser.parse_args(argv) - - runs = load_history(args.store) - - if args.list_runs: - for r in runs: - print(f"{r['timestamp']} {r['name']:<22} {r['run_hash']}") - return - - run = pick_run(runs, args.run) - chain = chain_of(runs, run["session_id"]) - graph = extract(args.store, run["session_id"]) - - known = set(graph["tasks"]) - resolved = [e for e in graph["edges"] if e["producer"] in known] - external = [e for e in graph["edges"] if e["producer"] == "EXTERNAL"] - dangling = [e for e in graph["edges"] - if e["producer"] not in known and e["producer"] != "EXTERNAL"] - stale = [t for t in graph["tasks"].values() if t.get("superseded")] - - print(f"run : {run['name']} ({run['run_hash'][:8]}, {run['timestamp']})") - print(f"session : {run['session_id']}") - if len(chain) > 1: - print(f" resume chain : {len(chain)} runs, " - f"{', '.join(r['name'] for r in chain)}") - print(f"tasks in session : {len(graph['tasks'])}") - if stale: - print(f" superseded : {len(stale)}") - print(f"input files (edges): {len(graph['edges'])}") - print(f" resolved to task : {len(resolved)}") - print(f" external inputs : {len(external)}") - print(f" DANGLING : {len(dangling)}") - - if dangling: - # A producer outside this session: another pipeline sharing the - # store, or a store pruned since the run. - print("\n=== DANGLING (producer not a task in this session) ===") - for e in dangling[:10]: - print(f"{e['consumer']} <- {e['producer']} ({e['filename']})") - - print("\n=== what this graph does not cover ===") - for note in graph["coverage"]: - print(f" - {note}") - - if args.json_out: - Path(args.json_out).write_text(json.dumps(graph, indent=2)) - print(f"\nwrote {args.json_out}") +class NextflowStore(Extractor): + name = "nextflow" + description = "a Nextflow .lineage store, including Seqera Platform" + + def add_arguments(self, parser): + parser.add_argument("--store", required=True, help="path to the .lineage directory") + parser.add_argument("--run", help="run name, run-hash prefix, or sessionId prefix " + "(default: most recent run)") + parser.add_argument("--list-runs", action="store_true", + help="list recorded runs and exit") + + def records(self, path): + path = Path(path) + if (path / ".lineage").is_dir(): + path = path / ".lineage" + if not (path / ".history").is_dir(): + return None + return {"root": path, + "runs": [{"name": r["name"], "id": r["run_hash"], "timestamp": r["timestamp"], + "session": r["session_id"]} for r in load_history(path)]} + + def load(self, root, run_id): + run = pick_run(load_history(root), run_id) + return extract(root, run["session_id"]) + + def extract(self, args): + runs = load_history(args.store) + if args.list_runs: + for r in runs: + print(f"{r['timestamp']} {r['name']:<22} {r['run_hash']}") + return None + self.run_record = pick_run(runs, args.run) + self.chain = chain_of(runs, self.run_record["session_id"]) + return extract(args.store, self.run_record["session_id"]) + + def summarize(self, graph, args): + run, chain = self.run_record, self.chain + known = set(graph["tasks"]) + resolved = [e for e in graph["edges"] if e["producer"] in known] + external = [e for e in graph["edges"] if e["producer"] == "EXTERNAL"] + dangling = [e for e in graph["edges"] + if e["producer"] not in known and e["producer"] != "EXTERNAL"] + stale = [t for t in graph["tasks"].values() if t.get("superseded")] + + print(f"run : {run['name']} ({run['run_hash'][:8]}, {run['timestamp']})") + print(f"session : {run['session_id']}") + if len(chain) > 1: + print(f" resume chain : {len(chain)} runs, " + f"{', '.join(r['name'] for r in chain)}") + print(f"tasks in session : {len(graph['tasks'])}") + if stale: + print(f" superseded : {len(stale)}") + print(f"input files (edges): {len(graph['edges'])}") + print(f" resolved to task : {len(resolved)}") + print(f" external inputs : {len(external)}") + print(f" DANGLING : {len(dangling)}") + if dangling: + # Another pipeline sharing the store, or a store pruned since the run. + print("\n=== DANGLING (producer not a task in this session) ===") + for e in dangling[:10]: + print(f"{e['consumer']} <- {e['producer']} ({e['filename']})") + self.coverage(graph) + + +main = NextflowStore.main if __name__ == "__main__": - main() + sys.exit(main()) diff --git a/clew/extract/rocrate.py b/providers/clew-nextflow/clew/provider/nextflow/extractor_rocrate.py similarity index 53% rename from clew/extract/rocrate.py rename to providers/clew-nextflow/clew/provider/nextflow/extractor_rocrate.py index e06d67d..1016858 100644 --- a/clew/extract/rocrate.py +++ b/providers/clew-nextflow/clew/provider/nextflow/extractor_rocrate.py @@ -1,54 +1,22 @@ """ -Clew — lineage adapter for Workflow Run RO-Crate (nf-prov output). - -WHY THIS EXISTS ---------------- -nf-prov publishes a run's provenance as a Workflow Run RO-Crate: a -standards-shaped JSON-LD file many archives, journals and registries -already ask for. Labs that produce crates for compliance reasons have -lineage on disk without knowing it — this adapter turns a crate into the -same graph JSON the other two extractors emit, so everything downstream -works unchanged. - -THE CRATE, AS WRITTEN BY nf-prov (wrroc format) ------------------------------------------------ - @graph contains, among much else: - CreateAction @id "#task/<32-hex>" one per task - name "ALIGN (sample_beta)" the nf-core tag convention - object[] inputs: "#task//" - or, for externals, whatever id the - file arrived under: file:///path, - https://... URLs, relative results/ - paths, #tmp entries. Verified against - a real nf-prov crate; only #param - references are not artifacts. - result[] outputs: "#task//" - instrument -> SoftwareApplication (container), when run - with one - CreateAction "Nextflow workflow run " run-level; skipped - - Note the instrument: in real crates it names the MODULE (a - SoftwareApplication like "bwa_index" with a source URL), not a container - image. No image appears anywhere in a crate, so version-pinned container - triggers ("gatk4:4.2.1") cannot match a crate graph; module-name matching - still works. - -Task hashes are the same 32-hex values the lineage store uses, abbreviated -identically, so graphs from either source are comparable node for node. - -HONEST LIMITS -------------- -A crate records what ran, not how to run it again: no script, and no -workdir. Tasks whose container is not in the crate therefore classify as -IRREDUCIBLE (fail closed — the crate carries no re-execution evidence), -and storage checks report DESTROYED unless published copies are mapped. -Prefer the lineage store when both exist; read the crate when it is what -a lab already has. +Workflow Run RO-Crates, as nf-prov writes them, read into the graph. + +In @graph each task is a CreateAction with @id "#task/<32-hex>", a name in +the "PROCESS (tag)" convention, object[] inputs as "#task//" +or an external id (file://, https://, results/ paths, #tmp), result[] +outputs, and an instrument naming the module, not a container image. +Version-pinned container triggers therefore cannot match; module names can. +Hashes abbreviate like the lineage store. + +A crate records what ran, not how to run it again: no script, no workdir. +Tasks without a container classify IRREDUCIBLE, and storage reads DESTROYED +unless published copies are mapped. Prefer the store when both exist. """ -import argparse import json +import sys from pathlib import Path +from clew.contracts import Extractor TASK_PREFIX = "#task/" FILE_SCHEME = "file://" @@ -127,7 +95,7 @@ def extract(crate_path): # whatever id the file arrived under: file:// paths, https:// # URLs for remote test data or references, relative results/ # paths, or #tmp entries. Dropping the unrecognised ones is - # how the first real crate lost all 117 external inputs — a + # how the first real crate lost all 117 external inputs, a # reference-update trigger then finds nothing and reads as # clean, which is the false negative this project exists to # avoid. Unknown ids fail open to EXTERNAL instead. @@ -149,28 +117,19 @@ def extract(crate_path): return {"tasks": tasks, "edges": edges, "outputs": outputs} -def main(argv=None): - parser = argparse.ArgumentParser( - description="Build a Clew graph from a Workflow Run RO-Crate (nf-prov).") - parser.add_argument("--crate", required=True, help="ro-crate-metadata.json path") - parser.add_argument("--json-out", help="path to write the graph as JSON") - args = parser.parse_args(argv) +class RoCrate(Extractor): + name = "ro-crate" + description = "a Workflow Run RO-Crate, as nf-prov writes" - graph = extract(args.crate) - known = set(graph["tasks"]) - external = [e for e in graph["edges"] if e["producer"] == "EXTERNAL"] - dangling = [e for e in graph["edges"] - if e["producer"] not in known and e["producer"] != "EXTERNAL"] + def add_arguments(self, parser): + parser.add_argument("--crate", required=True, help="ro-crate-metadata.json path") - print(f"tasks in crate : {len(graph['tasks'])}") - print(f"input files (edges): {len(graph['edges'])}") - print(f" external inputs : {len(external)}") - print(f" DANGLING : {len(dangling)}") + def extract(self, args): + return extract(args.crate) - if args.json_out: - Path(args.json_out).write_text(json.dumps(graph, indent=2)) - print(f"\nwrote {args.json_out}") + +main = RoCrate.main if __name__ == "__main__": - main() + sys.exit(main()) diff --git a/clew/extract/nextflow_work.py b/providers/clew-nextflow/clew/provider/nextflow/extractor_work.py similarity index 50% rename from clew/extract/nextflow_work.py rename to providers/clew-nextflow/clew/provider/nextflow/extractor_work.py index ccfb6c2..1f03d5a 100644 --- a/clew/extract/nextflow_work.py +++ b/providers/clew-nextflow/clew/provider/nextflow/extractor_work.py @@ -1,72 +1,25 @@ """ -Clew — lineage extractor for Nextflow runs. - -WHY THIS EXISTS ---------------- -Nextflow runs each pipeline step in its own private folder under work/. -A step's *input* files are not copied into that folder — that would waste -enormous disk space on genomic data. Instead Nextflow leaves a symlink -pointing at the file where it actually lives, which is the folder of the -step that produced it. - -Those symlinks were never intended as a provenance system. They exist to -save disk. But they accidentally record the complete history of the run: - - step P's folder contains: reads.bam -> work//reads.bam - => "P consumed a file that N produced" - -This script reads those symlinks and rebuilds the graph. - -WHAT IT PRODUCES ----------------- -An edge list, one line per input file: - - 13/46f32e <- fc/861a98 (chr22_1-40001.bed.gz) - 13/46f32e <- EXTERNAL (genome.fasta) - -Edges point BACKWARDS (consumer <- producer), because that is the direction -the filesystem records. Traversing downstream means inverting them. - -USAGE ------ - clew extract-work \ - --jsonl /path/to/Petri/logs/.jsonl \ - --work /path/to/Petri/work - -HONEST LIMITS -------------- -1. STAGING MODE. The trail exists only where inputs were staged as symlinks - (stageInMode 'symlink' or 'rellink', the default on local and HPC). Runs - staged by copy or hard link, including cloud executors pulling from - object storage, leave nothing to read. main() refuses rather than - returning an empty graph that would report every task as clean. - -2. PASS-THROUGH FILES LOSE THEIR PRODUCER. When a task re-emits an input - unchanged, Nextflow stages that file for the next task by pointing at - the ORIGINAL, not at the intermediate task's copy. On disk the hop - simply is not there: - - QUARTONOTEBOOK report.qmd -> ~/.nextflow/assets/.../report.qmd - MAKE_REPORT report.qmd -> ~/.nextflow/assets/.../report.qmd - - Both point at the asset, so this extractor calls the second one - EXTERNAL. The lineage store, which records channel lineage rather than - filesystem layout, attributes it to QUARTONOTEBOOK. - - Files a task genuinely produced are unaffected. The loss is confined to - files it forwarded untouched, and it is a FALSE NEGATIVE: an edge that - exists logically is missing from the graph, so a task fed only by - pass-throughs would look externally fed and its upstream would not be - reached. Prefer the lineage store where pass-through is common. +Nextflow work/ symlinks, read into the graph. + +Each task runs in its own folder under work/, and its inputs are symlinks +into the producing task's folder. That was meant to save disk; it also +records the run. Edges are stored consumer <- producer, the direction the +filesystem holds them. + +Limits. Only symlink staging leaves a trail: copy, hard link and cloud +executors do not, and main() refuses rather than emit an empty graph. A file +a task forwards unchanged is staged from the original, so that hop is +missing and the next task looks externally fed. Prefer the lineage store +where pass-through is common. """ -import argparse import json import os import sys from pathlib import Path from clew.graph.graph import task_status +from clew.contracts import Extractor # Nextflow's own bookkeeping files. Not data, never lineage. SKIP_PREFIXES = (".command", ".exitcode") @@ -117,16 +70,10 @@ def load_run(jsonl_path): def find_workdir(work_root, task_hash): """ - Turn a task hash like 'fc/861a98' into its folder on disk. - - The hash is abbreviated: two-character directory, then the first six - characters of a much longer directory name. So we list the parent and - match by prefix. - - Two directories sharing the prefix cannot be told apart by the hash, - and every symlink into either would be credited to one task. That is - refused rather than guessed; it takes on the order of a million task - directories to happen, so the refusal costs nothing until it matters. + The folder on disk for a hash like fc/861a98: list the two-character + directory and match the six-character prefix. Two folders sharing the + prefix are refused rather than guessed; that takes about a million task + directories. """ prefix_dir, name_prefix = task_hash.split("/", 1) parent = Path(work_root) / prefix_dir @@ -145,23 +92,10 @@ def find_workdir(work_root, task_hash): def target_to_hash(target, work_root): """ - Given the path a symlink points at, work out which task produced it. - - Returns one of: - "XX/YYYYYY" -- a task hash - "EXTERNAL" -- came from outside the pipeline - None -- unrecognisable - - THE TRAP - -------- - Files entering the pipeline from outside are staged under directories - named 'stage-', like: - - work/stage-b3809b93-.../3e/d2a534.../genome.fasta - - That '3e/d2a534' looks EXACTLY like a task hash and is not one. Checking - for 'stage-' must happen before any hash parsing, or the graph fills up - with tasks that never existed. + Which task produced the file a symlink points at: "XX/YYYYYY", + "EXTERNAL", or None. External inputs are staged under + work/stage-/3e/d2a534.../, which looks exactly like a task hash + and is not one, so the stage- check runs before any hash parsing. """ target = Path(target) work_root = Path(work_root).resolve() @@ -272,7 +206,7 @@ def extract(jsonl_path, work_root): # Size is recorded here, while the workdir still exists, because it # is the only join key that survives publishDir copying a file: the # copy keeps its name and byte count but gets a new path and mtime. - # Read it now or lose it — this extractor runs against a work tree + # Read it now or lose it, this extractor runs against a work tree # that is about to be cleaned. details = [] for path in produced: @@ -303,94 +237,86 @@ def coverage_notes(missing): return notes -def main(argv=None): - parser = argparse.ArgumentParser(description="Rebuild Nextflow lineage from work/ symlinks.") - parser.add_argument("--jsonl", required=True, help="Petri weblog JSONL for one run") - parser.add_argument("--work", required=True, help="Nextflow work/ directory") - parser.add_argument("--json-out", help="Optional path to write the graph as JSON") - parser.add_argument("--allow-partial", action="store_true", - help="write the graph even when some tasks' work " - "directories are gone; the missing tasks are " - "recorded as a coverage note on the graph") - args = parser.parse_args(argv) - - tasks, edges, outputs, output_details, missing = extract( - args.jsonl, args.work) - - # A task whose directory is gone contributes no edges, so everything - # it fed looks externally fed and a withdrawal stops short of it. That - # is a false negative, and the default is to refuse rather than write - # a graph that under-reports. --allow-partial writes it, with the gap - # recorded on the graph itself. - if missing and not args.allow_partial: - sys.exit( - f"clew extract-work: {len(missing)} of {len(tasks)} tasks have " - f"no work directory under {args.work}:\n " - + "\n ".join(missing) - + "\nEither the work tree was cleaned, or --work names another " - "run's tree. A graph missing these tasks would under-report; " - "pass --allow-partial to write it anyway with the gap recorded " - "as a coverage note, or use the lineage store if the run " - "recorded one.") - - # A work tree with tasks but no symlinks at all means the inputs were - # staged by copy or hard link (stageInMode 'copy'/'link', or a cloud - # executor downloading from object storage). There is no lineage to read - # here, and returning an empty graph would report every task as having - # no inputs - a false negative dressed up as an answer. Refuse instead. - if tasks and not edges and len(missing) < len(tasks): - sys.exit( - "clew extract-work: no symlinks found in any task directory.\n" - "This extractor reads the symlinks Nextflow leaves with the\n" - "default stage-in mode ('symlink'/'rellink') on local and HPC\n" - "executors. Runs staged by copy or hard link, including cloud\n" - "executors staging from object storage, leave no symlinks to\n" - "read. For those runs, enable Nextflow's lineage store and use\n" - "'clew extract-store' instead.") - - known = set(tasks) - resolved = [e for e in edges if e["producer"] in known] - external = [e for e in edges if e["producer"] == "EXTERNAL"] - dangling = [e for e in edges if e["producer"] not in known and e["producer"] != "EXTERNAL"] - - print(f"tasks in run : {len(tasks)}") - print(f"work dirs missing : {len(missing)}") - for task_hash in missing: - print(f" {task_hash} {tasks[task_hash].get('name', '')}") - print(f"input files (edges): {len(edges)}") - print(f" resolved to task : {len(resolved)}") - print(f" external inputs : {len(external)}") - print(f" DANGLING : {len(dangling)}") - print() - - print("=== EDGES (consumer <- producer) ===") - for e in edges: - producer = e["producer"] or "???" - print(f"{e['consumer']} <- {producer:<12} ({e['filename']})") - - if dangling: - # Dangling edges almost always mean the stage- check failed, or the - # work/ directory belongs to a different run than the JSONL. - print() - print("=== DANGLING (producer not a task in this run) ===") - for e in dangling: - print(f"{e['consumer']} <- {e['producer']} ({e['filename']})") - print(f" target: {e['target']}") - - coverage = coverage_notes(missing) - if coverage: - print("\n=== what this graph does not cover ===") - for note in coverage: - print(f" - {note}") - - if args.json_out: +class NextflowWork(Extractor): + name = "nextflow-work" + description = "a Nextflow work/ tree, from its symlinks, any engine version" + + def add_arguments(self, parser): + parser.add_argument("--jsonl", required=True, help="Petri weblog JSONL for one run") + parser.add_argument("--work", required=True, help="Nextflow work/ directory") + parser.add_argument("--allow-partial", action="store_true", + help="write the graph even when some tasks' work " + "directories are gone; the missing tasks are " + "recorded as a coverage note on the graph") + + def extract(self, args): + tasks, edges, outputs, output_details, missing = extract(args.jsonl, args.work) + self.missing = missing + + # A task whose directory is gone contributes no edges, so a withdrawal + # stops short of everything it fed. Refuse rather than under-report. + if missing and not args.allow_partial: + sys.exit( + f"clew extract nextflow-work: {len(missing)} of {len(tasks)} tasks have " + f"no work directory under {args.work}:\n " + + "\n ".join(missing) + + "\nEither the work tree was cleaned, or --work names another " + "run's tree. A graph missing these tasks would under-report; " + "pass --allow-partial to write it anyway with the gap recorded " + "as a coverage note, or use the lineage store if the run " + "recorded one.") + + # Tasks but no symlinks: inputs were staged by copy or hard link, so + # there is no lineage here. An empty graph would be a false negative. + if tasks and not edges and len(missing) < len(tasks): + sys.exit( + "clew extract nextflow-work: no symlinks found in any task directory.\n" + "This extractor reads the symlinks Nextflow leaves with the\n" + "default stage-in mode ('symlink'/'rellink') on local and HPC\n" + "executors. Runs staged by copy or hard link, including cloud\n" + "executors staging from object storage, leave no symlinks to\n" + "read. For those runs, enable Nextflow's lineage store and use\n" + "'clew extract nextflow' instead.") + graph = {"tasks": tasks, "edges": edges, "outputs": outputs, "output_details": output_details} + coverage = coverage_notes(missing) if coverage: graph["coverage"] = coverage - Path(args.json_out).write_text(json.dumps(graph, indent=2)) - print(f"\nwrote {args.json_out}") + return graph + + def summarize(self, graph, args): + tasks, edges = graph["tasks"], graph["edges"] + known = set(tasks) + resolved = [e for e in edges if e["producer"] in known] + external = [e for e in edges if e["producer"] == "EXTERNAL"] + dangling = [e for e in edges if e["producer"] not in known and e["producer"] != "EXTERNAL"] + + print(f"tasks in run : {len(tasks)}") + print(f"work dirs missing : {len(self.missing)}") + for task_hash in self.missing: + print(f" {task_hash} {tasks[task_hash].get('name', '')}") + print(f"input files (edges): {len(edges)}") + print(f" resolved to task : {len(resolved)}") + print(f" external inputs : {len(external)}") + print(f" DANGLING : {len(dangling)}") + print() + print("=== EDGES (consumer <- producer) ===") + for e in edges: + producer = e["producer"] or "???" + print(f"{e['consumer']} <- {producer:<12} ({e['filename']})") + if dangling: + # Almost always the stage- check failed, or work/ belongs to another run. + print() + print("=== DANGLING (producer not a task in this run) ===") + for e in dangling: + print(f"{e['consumer']} <- {e['producer']} ({e['filename']})") + print(f" target: {e['target']}") + self.coverage(graph) + + +main = NextflowWork.main if __name__ == "__main__": - main() + sys.exit(main()) diff --git a/providers/clew-nextflow/pyproject.toml b/providers/clew-nextflow/pyproject.toml new file mode 100644 index 0000000..d748373 --- /dev/null +++ b/providers/clew-nextflow/pyproject.toml @@ -0,0 +1,28 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "clew-nextflow" +version = "0.4.0" +description = "Clew provider for Nextflow: the .lineage store, work/ symlinks, Workflow Run RO-Crates, and the nf-core pipeline adapters" +readme = "README.md" +requires-python = ">=3.9" +license = { text = "AGPL-3.0-only" } +authors = [{ name = "QuietFlare" }] +dependencies = ["clew-lineage>=0.4"] + +[project.entry-points."clew.extractors"] +nextflow = "clew.provider.nextflow.extractor_lineage" +nextflow-work = "clew.provider.nextflow.extractor_work" +ro-crate = "clew.provider.nextflow.extractor_rocrate" + +[project.entry-points."clew.adapters"] +sarek = "clew.provider.nextflow.adapter_sarek" +rnaseq = "clew.provider.nextflow.adapter_rnaseq" +viralrecon = "clew.provider.nextflow.adapter_viralrecon" + +[tool.setuptools.packages.find] +# Installs into the clew.provider namespace; no __init__.py above the provider. +include = ["clew*"] +namespaces = true diff --git a/tests/fixtures/ro-crate-metadata.json b/providers/clew-nextflow/tests/fixtures/ro-crate-metadata.json similarity index 100% rename from tests/fixtures/ro-crate-metadata.json rename to providers/clew-nextflow/tests/fixtures/ro-crate-metadata.json diff --git a/tests/fixtures/rocrate_samples.csv b/providers/clew-nextflow/tests/fixtures/rocrate_samples.csv similarity index 100% rename from tests/fixtures/rocrate_samples.csv rename to providers/clew-nextflow/tests/fixtures/rocrate_samples.csv diff --git a/providers/clew-nextflow/tests/test_drift_store.py b/providers/clew-nextflow/tests/test_drift_store.py new file mode 100644 index 0000000..d407463 --- /dev/null +++ b/providers/clew-nextflow/tests/test_drift_store.py @@ -0,0 +1,42 @@ +""" +Drift pairs tasks by name across two runs, compares output digests, and +names the first task on each chain that differs together with the cause +read from the record. Every verdict is pinned on a small synthetic chain. +""" + +import sys +import unittest +from pathlib import Path + + +from clew.questions import drift + + +class SameChain(unittest.TestCase): + """ + The store holds one graph per resume chain. Two runs of one chain + load the same graph, and drift compared it with itself. + """ + + def test_two_runs_of_one_session_are_refused(self): + import contextlib + import io + import tempfile + from test_lineage_store import RUN_A, RUN_B, CHAIN, PRODUCER, task_run, write_record + with tempfile.TemporaryDirectory() as tmp: + store = Path(tmp) + (store / ".history").mkdir() + (store / ".history" / RUN_A).write_text( + f"2026-08-01 10:00:00 CEST\tfirst_run\t{CHAIN}\tlid://{RUN_A}\n") + (store / ".history" / RUN_B).write_text( + f"2026-08-02 10:00:00 CEST\tsecond_run\t{CHAIN}\tlid://{RUN_B}\n") + write_record(store / PRODUCER, task_run(CHAIN, RUN_A, "PIPE:ALIGN")) + with self.assertRaises(SystemExit) as refused, \ + contextlib.redirect_stdout(io.StringIO()): + drift.main(["--runs", tmp, "--before", "first_run", "--after", "second_run"]) + self.assertIn("same graph", str(refused.exception)) + self.assertIn("resume chain", str(refused.exception)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_extractor.py b/providers/clew-nextflow/tests/test_extractor.py similarity index 95% rename from tests/test_extractor.py rename to providers/clew-nextflow/tests/test_extractor.py index 9e2e20c..3ecb6bb 100644 --- a/tests/test_extractor.py +++ b/providers/clew-nextflow/tests/test_extractor.py @@ -1,6 +1,6 @@ """ The extractor traps that cost real time. Each test here is a regression -guard for a bug that produced FALSE NEGATIVES — the graph reporting donor +guard for a bug that produced FALSE NEGATIVES, the graph reporting donor data as absent when it was present. """ @@ -12,9 +12,9 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.extract import nextflow_work as ex +from clew.provider.nextflow import extractor_work as ex class TestTargetToHash(unittest.TestCase): @@ -49,7 +49,7 @@ class TestNumberedSubdirectories(unittest.TestCase): """ Trap 3: aggregating tasks stage inputs in numbered subdirectories (./1/, ./18/). Walking only the top level made MULTIQC appear to have - 1 input instead of 20 — hiding exactly the many-into-one nodes this + 1 input instead of 20, hiding exactly the many-into-one nodes this project exists to track. """ @@ -151,13 +151,13 @@ def tearDown(self): def test_cli_exits_nonzero_and_points_at_the_lineage_store(self): result = subprocess.run( - [sys.executable, "-m", "clew.extract.nextflow_work", + [sys.executable, "-m", "clew.provider.nextflow.extractor_work", "--jsonl", str(self.jsonl), "--work", str(self.work)], capture_output=True, text=True, - cwd=Path(__file__).resolve().parent.parent) + cwd=Path(__file__).resolve().parents[3]) self.assertNotEqual(result.returncode, 0) self.assertIn("no symlinks", result.stderr) - self.assertIn("extract-store", result.stderr) + self.assertIn("clew extract nextflow", result.stderr) class TestMissingWorkDirectoriesAreRefused(unittest.TestCase): """ @@ -185,11 +185,11 @@ def tearDown(self): def run_cli(self, *extra): return subprocess.run( - [sys.executable, "-m", "clew.extract.nextflow_work", + [sys.executable, "-m", "clew.provider.nextflow.extractor_work", "--jsonl", str(self.jsonl), "--work", str(self.work), "--json-out", str(self.out), *extra], capture_output=True, text=True, - cwd=Path(__file__).resolve().parent.parent) + cwd=Path(__file__).resolve().parents[3]) def test_refused_by_default_naming_the_tasks(self): result = self.run_cli() @@ -231,7 +231,7 @@ class TestOutputSizesAreRecorded(unittest.TestCase): """ Sizes must be read while the work tree still exists. A published copy keeps its name and byte count but gets a new path and mtime, so - (basename, size) is the only join key that survives publishDir — and + (basename, size) is the only join key that survives publishDir, and the workdir is usually deleted soon after the run. """ diff --git a/tests/test_graph5_regression.py b/providers/clew-nextflow/tests/test_graph5_regression.py similarity index 80% rename from tests/test_graph5_regression.py rename to providers/clew-nextflow/tests/test_graph5_regression.py index c14f456..f483ba6 100644 --- a/tests/test_graph5_regression.py +++ b/providers/clew-nextflow/tests/test_graph5_regression.py @@ -10,12 +10,13 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) from clew.graph import blast_radius as core -from clew.domains import sarek +from clew.graph import graph as core_graph +from clew.provider.nextflow import adapter_sarek as sarek -ROOT = Path(__file__).resolve().parent.parent +ROOT = Path(__file__).resolve().parents[3] GRAPH = ROOT / "clew" / "data" / "graph5.json" SHEET = ROOT / "clew" / "data" / "donors.csv" @@ -25,8 +26,9 @@ class TestFiveDonorRun(unittest.TestCase): @classmethod def setUpClass(cls): cls.graph = core.load_graph(GRAPH) - cls.donors = sarek.load_donors(SHEET) - cls.entry = sarek.subject_entry_nodes(cls.graph, cls.donors) + kind = sarek.Sarek().triggers['patient'] + cls.donors = kind.ids(SHEET) + cls.entry = kind.entries(cls.graph, cls.donors) cls.radius = core.blast_radius(cls.graph, cls.entry) def test_shape(self): @@ -42,7 +44,7 @@ def test_no_dangling_producers(self): def test_every_donor_has_entry_points(self): # Zero entry points for a listed donor means attribution silently - # failed — the false-negative direction. + # failed, the false-negative direction. for donor, nodes in self.entry.items(): self.assertEqual(len(nodes), 15, donor) @@ -52,32 +54,32 @@ def test_withdrawal_radius_per_donor(self): self.assertEqual(len(r["exclusive"]), 15, donor) # The single shared node is the aggregator. (shared,) = r["shared"] - self.assertEqual(sarek.describe(self.graph, shared), "MULTIQC") + self.assertEqual(core_graph.describe(self.graph, shared), "MULTIQC") def test_aggregator_sees_every_donor(self): # MULTIQC must be reachable from all five donors. This was the bug # that reported 110+ donor-derived files as clean. for donor, r in self.radius.items(): self.assertTrue( - any(sarek.describe(self.graph, h) == "MULTIQC" + any(core_graph.describe(self.graph, h) == "MULTIQC" for h in r["affected"]), f"{donor} does not reach MULTIQC", ) def test_reference_update_is_load_bearing(self): - subjects = sarek.external_input_entry_nodes(self.graph, "genome.fasta") + subjects = core_graph.external_input_entry_nodes(self.graph, "genome.fasta") entry = subjects["input:genome.fasta"] self.assertEqual(len(entry), 41) radius = core.blast_radius(self.graph, subjects) affected = radius["input:genome.fasta"]["affected"] - # The reference reaches most of the run — far beyond any one donor. + # The reference reaches most of the run, far beyond any one donor. self.assertEqual(len(affected), 72) biggest_donor = max(len(r["affected"]) for r in self.radius.values()) self.assertGreater(len(affected), biggest_donor) def test_tool_defect_radius(self): - subjects = sarek.container_entry_nodes(self.graph, "gatk4") + subjects = core_graph.container_entry_nodes(self.graph, "gatk4") radius = core.blast_radius(self.graph, subjects) self.assertEqual(len(radius["container:gatk4"]["affected"]), 68) @@ -92,7 +94,7 @@ def test_a_versioned_needle_reaches_every_samtools_task(self): self.assertEqual(len(name_only), 11) self.assertTrue(all("_samtools" in self.graph["tasks"][h]["container"] for h in name_only)) - subjects = sarek.container_entry_nodes(self.graph, "samtools:1.21") + subjects = core_graph.container_entry_nodes(self.graph, "samtools:1.21") radius = core.blast_radius(self.graph, subjects) self.assertEqual(len(radius["container:samtools:1.21"]["affected"]), 72) @@ -100,7 +102,7 @@ def test_donor_reference_asymmetry(self): # Withdrawing a donor must NOT pull in the reference-only tasks: # one withdrawal must never poison every sample ever aligned. ref_entry = set( - sarek.external_input_entry_nodes(self.graph, "genome.fasta")["input:genome.fasta"] + core_graph.external_input_entry_nodes(self.graph, "genome.fasta")["input:genome.fasta"] ) for donor, r in self.radius.items(): self.assertLess(len(r["affected"] & ref_entry), len(ref_entry), donor) diff --git a/tests/test_graph_chain_regression.py b/providers/clew-nextflow/tests/test_graph_chain_regression.py similarity index 86% rename from tests/test_graph_chain_regression.py rename to providers/clew-nextflow/tests/test_graph_chain_regression.py index cde5d22..385d67f 100644 --- a/tests/test_graph_chain_regression.py +++ b/providers/clew-nextflow/tests/test_graph_chain_regression.py @@ -11,12 +11,12 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) from clew.graph import blast_radius as core -from clew.domains import rnaseq +from clew.provider.nextflow import adapter_rnaseq as rnaseq -ROOT = Path(__file__).resolve().parent.parent +ROOT = Path(__file__).resolve().parents[3] GRAPH = ROOT / "clew" / "data" / "graph_chain.json" SHEET = ROOT / "clew" / "data" / "samplesheets" / "rnaseq_yeast.csv" @@ -26,8 +26,9 @@ class TestCrossRunChain(unittest.TestCase): @classmethod def setUpClass(cls): cls.graph = core.load_graph(GRAPH) - cls.subjects = rnaseq.load_subjects(SHEET) - cls.entry = rnaseq.subject_entry_nodes(cls.graph, cls.subjects) + kind = rnaseq.Rnaseq().triggers['sample'] + cls.subjects = kind.ids(SHEET) + cls.entry = kind.entries(cls.graph, cls.subjects) cls.radius = core.blast_radius(cls.graph, cls.entry) def test_shape(self): @@ -50,7 +51,7 @@ def test_withdrawal_crosses_the_boundary(self): self.assertEqual(len(r["affected"]), 57) # 46 in rna + 11 in da da_reached = {h for h in r["affected"] if h.startswith("da:")} self.assertEqual(len(da_reached), 11) - # The report bundle — the artifact people actually receive — is in + # The report bundle, the artifact people actually receive, is in # the radius. This is the row that makes the plan legible. names = {self.graph["tasks"][h]["process"].rsplit(":", 1)[-1] for h in da_reached} @@ -58,8 +59,7 @@ def test_withdrawal_crosses_the_boundary(self): self.assertIn("DESEQ2_DIFFERENTIAL", names) def test_every_sample_reaches_the_de_run(self): - # The matrix mixes ALL samples, so every withdrawal crosses over — - # the many-into-one aggregation working across a run boundary. + # The matrix mixes ALL samples, so every withdrawal crosses over, # the many-into-one aggregation working across a run boundary. for sample, r in self.radius.items(): self.assertTrue(any(h.startswith("da:") for h in r["affected"]), sample) diff --git a/tests/test_graph_rna_regression.py b/providers/clew-nextflow/tests/test_graph_rna_regression.py similarity index 77% rename from tests/test_graph_rna_regression.py rename to providers/clew-nextflow/tests/test_graph_rna_regression.py index f50ca44..9fc0ae4 100644 --- a/tests/test_graph_rna_regression.py +++ b/providers/clew-nextflow/tests/test_graph_rna_regression.py @@ -4,19 +4,20 @@ extracted from the native lineage store (graph_rna.json). Third pipeline through the shared nfcore adapter machinery, zero core or -adapter-machinery changes — these numbers pin that claim. +adapter-machinery changes, these numbers pin that claim. """ import sys import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) from clew.graph import blast_radius as core -from clew.domains import rnaseq +from clew.graph import graph as core_graph +from clew.provider.nextflow import adapter_rnaseq as rnaseq -ROOT = Path(__file__).resolve().parent.parent +ROOT = Path(__file__).resolve().parents[3] GRAPH = ROOT / "clew" / "data" / "graph_rna.json" SHEET = ROOT / "clew" / "data" / "samplesheets" / "rnaseq_yeast.csv" @@ -26,8 +27,9 @@ class TestRnaseqRun(unittest.TestCase): @classmethod def setUpClass(cls): cls.graph = core.load_graph(GRAPH) - cls.subjects = rnaseq.load_subjects(SHEET) - cls.entry = rnaseq.subject_entry_nodes(cls.graph, cls.subjects) + kind = rnaseq.Rnaseq().triggers['sample'] + cls.subjects = kind.ids(SHEET) + cls.entry = kind.entries(cls.graph, cls.subjects) cls.radius = core.blast_radius(cls.graph, cls.entry) def test_shape(self): @@ -54,7 +56,7 @@ def test_sample_withdrawal_radius(self): self.assertEqual(len(r["shared"]), 5, sample) def test_gtf_shallow_entry_deep_reach(self): - subjects = rnaseq.external_input_entry_nodes(self.graph, "genes.gtf") + subjects = core_graph.external_input_entry_nodes(self.graph, "genes.gtf") entry = subjects["input:genes.gtf"] self.assertEqual(len(entry), 1) # one direct consumer... radius = core.blast_radius(self.graph, subjects) @@ -62,25 +64,24 @@ def test_gtf_shallow_entry_deep_reach(self): self.assertEqual(len(radius["input:genes.gtf"]["affected"]), 149) def test_genome_radius(self): - subjects = rnaseq.external_input_entry_nodes(self.graph, "genome.fa") + subjects = core_graph.external_input_entry_nodes(self.graph, "genome.fa") radius = core.blast_radius(self.graph, subjects) self.assertEqual(len(radius["input:genome.fa"]["affected"]), 150) def test_aligner_vs_quantifier_asymmetry(self): # A STAR defect poisons nearly everything; a Salmon defect stays - # contained. Same trigger type, order-of-magnitude different radius — - # the reason per-tool blast radii matter. + # contained. Same trigger type, order-of-magnitude different radius, # the reason per-tool blast radii matter. star = core.blast_radius(self.graph, - rnaseq.container_entry_nodes(self.graph, "star")) + core_graph.container_entry_nodes(self.graph, "star")) salmon = core.blast_radius(self.graph, - rnaseq.container_entry_nodes(self.graph, "salmon")) + core_graph.container_entry_nodes(self.graph, "salmon")) self.assertEqual(len(star["container:star"]["affected"]), 148) self.assertEqual(len(salmon["container:salmon"]["affected"]), 15) def test_every_load_bearing_input_present(self): seen = {Path(e["filename"]).name for e in self.graph["edges"] if e["producer"] == "EXTERNAL"} - for name in rnaseq.LOAD_BEARING_INPUTS: + for name in rnaseq.Rnaseq().load_bearing_inputs: self.assertIn(name, seen, name) diff --git a/tests/test_graph_vr_regression.py b/providers/clew-nextflow/tests/test_graph_vr_regression.py similarity index 79% rename from tests/test_graph_vr_regression.py rename to providers/clew-nextflow/tests/test_graph_vr_regression.py index 491ea57..890a8f2 100644 --- a/tests/test_graph_vr_regression.py +++ b/providers/clew-nextflow/tests/test_graph_vr_regression.py @@ -4,19 +4,20 @@ 219 tasks, extracted from the native lineage store (graph_vr.json). This is the first pipeline the nfcore adapter machinery never saw during -development — these numbers pin the proof that it generalised. +development, these numbers pin the proof that it generalised. """ import sys import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) from clew.graph import blast_radius as core -from clew.domains import viralrecon +from clew.graph import graph as core_graph +from clew.provider.nextflow import adapter_viralrecon as viralrecon -ROOT = Path(__file__).resolve().parent.parent +ROOT = Path(__file__).resolve().parents[3] GRAPH = ROOT / "clew" / "data" / "graph_vr.json" SHEET = ROOT / "clew" / "data" / "samplesheets" / "viralrecon_coguk.csv" @@ -26,8 +27,9 @@ class TestViralreconRun(unittest.TestCase): @classmethod def setUpClass(cls): cls.graph = core.load_graph(GRAPH) - cls.subjects = viralrecon.load_subjects(SHEET) - cls.entry = viralrecon.subject_entry_nodes(cls.graph, cls.subjects) + kind = viralrecon.Viralrecon().triggers['sample'] + cls.subjects = kind.ids(SHEET) + cls.entry = kind.entries(cls.graph, cls.subjects) cls.radius = core.blast_radius(cls.graph, cls.entry) def test_shape(self): @@ -43,7 +45,7 @@ def test_no_dangling_producers(self): def test_every_specimen_attributed(self): # 41 entry tasks per specimen; zero would mean the tag parser - # silently failed on this pipeline's naming — the false-negative + # silently failed on this pipeline's naming, the false-negative # direction. for specimen, nodes in self.entry.items(): self.assertEqual(len(nodes), 41, specimen) @@ -55,7 +57,7 @@ def test_specimen_withdrawal_radius(self): self.assertEqual(len(r["shared"]), 5, specimen) def test_primer_scheme_is_load_bearing(self): - subjects = viralrecon.external_input_entry_nodes( + subjects = core_graph.external_input_entry_nodes( self.graph, "nCoV-2019.primer.bed") entry = subjects["input:nCoV-2019.primer.bed"] self.assertEqual(len(entry), 11) @@ -63,14 +65,14 @@ def test_primer_scheme_is_load_bearing(self): self.assertEqual(len(radius["input:nCoV-2019.primer.bed"]["affected"]), 161) def test_reference_reaches_furthest(self): - subjects = viralrecon.external_input_entry_nodes( + subjects = core_graph.external_input_entry_nodes( self.graph, "nCoV-2019.reference.fasta") radius = core.blast_radius(self.graph, subjects) self.assertEqual( len(radius["input:nCoV-2019.reference.fasta"]["affected"]), 193) def test_ivar_defect_radius(self): - subjects = viralrecon.container_entry_nodes(self.graph, "ivar") + subjects = core_graph.container_entry_nodes(self.graph, "ivar") radius = core.blast_radius(self.graph, subjects) self.assertEqual(len(radius["container:ivar"]["affected"]), 160) @@ -80,7 +82,7 @@ def test_every_load_bearing_input_present(self): from pathlib import Path as P seen = {P(e["filename"]).name for e in self.graph["edges"] if e["producer"] == "EXTERNAL"} - for name in viralrecon.LOAD_BEARING_INPUTS: + for name in viralrecon.Viralrecon().load_bearing_inputs: self.assertIn(name, seen, name) diff --git a/tests/test_lineage_store.py b/providers/clew-nextflow/tests/test_lineage_store.py similarity index 94% rename from tests/test_lineage_store.py rename to providers/clew-nextflow/tests/test_lineage_store.py index fb2d71f..a28b7a0 100644 --- a/tests/test_lineage_store.py +++ b/providers/clew-nextflow/tests/test_lineage_store.py @@ -18,9 +18,9 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.extract import nextflow_store as ls +from clew.provider.nextflow import extractor_lineage as ls RUN_A = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" RUN_B = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" @@ -246,7 +246,7 @@ def test_workpath_is_the_hash_split_as_the_work_tree_splits_it(self): f"{PRODUCER[:2]}/{PRODUCER[2:]}") def test_tag_preserved_for_domain_parsing(self): - # Domain adapters attribute subjects by parsing "(tag)" from the + # Adapter adapters attribute subjects by parsing "(tag)" from the # name; the adapter must not strip it from `name`, only `process`. graph = ls.extract(self.store, CHAIN) task = graph["tasks"][ls.abbreviate(PRODUCER)] @@ -321,7 +321,7 @@ def test_an_unread_record_kind_is_reported(self): self.assertTrue(any("AgentRun" in n for n in graph["coverage"])) -PETRI_STORE = Path(__file__).resolve().parent.parent.parent / "Petri" / ".lineage" +PETRI_STORE = Path(__file__).resolve().parents[4] / "Petri" / ".lineage" @unittest.skipUnless(PETRI_STORE.is_dir(), "Petri lineage store not present") @@ -334,15 +334,17 @@ class RealStoreEquivalence(unittest.TestCase): @classmethod def setUpClass(cls): - sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + sys.path.insert(0, str(Path(__file__).resolve().parents[3])) from clew.graph import blast_radius as core - from clew.domains import sarek + from clew.graph import graph as core_graph + from clew.provider.nextflow import adapter_sarek as sarek runs = ls.load_history(PETRI_STORE) run = ls.pick_run(runs, "tender_mccarthy") cls.graph = ls.extract(PETRI_STORE, run["session_id"]) cls.core = core cls.sarek = sarek + cls.core_graph = core_graph def test_shape(self): self.assertEqual(len(self.graph["tasks"]), 81) @@ -354,24 +356,24 @@ def test_no_dangling(self): self.assertEqual(dangling, []) def test_withdrawal_numbers_match_symlink_extractor(self): - donors = self.sarek.load_donors( - Path(__file__).resolve().parent.parent / "clew" / "data" / "donors.csv") - entry = self.sarek.subject_entry_nodes(self.graph, donors) + kind = self.sarek.Sarek().triggers["patient"] + donors = kind.ids(Path(__file__).resolve().parents[3] / "clew" / "data" / "donors.csv") + entry = kind.entries(self.graph, donors) radius = self.core.blast_radius(self.graph, entry) for donor, r in radius.items(): self.assertEqual(len(entry[donor]), 15, donor) self.assertEqual(len(r["affected"]), 16, donor) self.assertEqual(len(r["exclusive"]), 15, donor) (shared,) = r["shared"] - self.assertEqual(self.sarek.describe(self.graph, shared), "MULTIQC") + self.assertEqual(self.core_graph.describe(self.graph, shared), "MULTIQC") def test_reference_and_container_numbers_match(self): - subjects = self.sarek.external_input_entry_nodes(self.graph, "genome.fasta") + subjects = self.core_graph.external_input_entry_nodes(self.graph, "genome.fasta") self.assertEqual(len(subjects["input:genome.fasta"]), 41) radius = self.core.blast_radius(self.graph, subjects) self.assertEqual(len(radius["input:genome.fasta"]["affected"]), 72) - subjects = self.sarek.container_entry_nodes(self.graph, "gatk4") + subjects = self.core_graph.container_entry_nodes(self.graph, "gatk4") radius = self.core.blast_radius(self.graph, subjects) self.assertEqual(len(radius["container:gatk4"]["affected"]), 68) diff --git a/providers/clew-nextflow/tests/test_nextflow_contract.py b/providers/clew-nextflow/tests/test_nextflow_contract.py new file mode 100644 index 0000000..7a0292b --- /dev/null +++ b/providers/clew-nextflow/tests/test_nextflow_contract.py @@ -0,0 +1,29 @@ +"""What this extractor emits is the one graph every question reads.""" + +import unittest +from pathlib import Path + +from clew.graph.graph import STATUSES, contract_violations +from clew.provider.nextflow import extractor_rocrate as rc + +FIXTURES = Path(__file__).resolve().parent / "fixtures" + + +class Contract(unittest.TestCase): + def graphs(self): + yield "ro-crate", rc.extract(FIXTURES / "ro-crate-metadata.json") + + def test_conforms(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + self.assertEqual(contract_violations(graph), []) + + def test_statuses_are_in_the_vocabulary(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + for task in graph["tasks"].values(): + self.assertIn(task["status"], STATUSES) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_rocrate.py b/providers/clew-nextflow/tests/test_rocrate.py similarity index 83% rename from tests/test_rocrate.py rename to providers/clew-nextflow/tests/test_rocrate.py index 9baab79..7090790 100644 --- a/tests/test_rocrate.py +++ b/providers/clew-nextflow/tests/test_rocrate.py @@ -3,7 +3,7 @@ (fixtures/ro-crate-metadata.json): a 5-task run with two samples, one shared reference, and an aggregator. -Third ingest path, same graph schema, same engine — plus the honest +Third ingest path, same graph schema, same engine, plus the honest consequences of what a crate does NOT record. """ @@ -13,11 +13,13 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.extract import rocrate as rc +from clew.provider.nextflow import extractor_rocrate as rc from clew.graph import blast_radius as core -from clew.domains import viralrecon # sample-keyed adapter; the crate has no donors +from clew.graph import graph as core_graph +from clew.graph import contribution +from clew.provider.nextflow import adapter_viralrecon as viralrecon # sample-keyed adapter; the crate has no donors FIXTURES = Path(__file__).resolve().parent / "fixtures" CRATE = FIXTURES / "ro-crate-metadata.json" @@ -29,8 +31,9 @@ class TestRoCrateAdapter(unittest.TestCase): @classmethod def setUpClass(cls): cls.graph = rc.extract(CRATE) - cls.subjects = viralrecon.load_subjects(SHEET) - cls.entry = viralrecon.subject_entry_nodes(cls.graph, cls.subjects) + kind = viralrecon.Viralrecon().triggers['sample'] + cls.subjects = kind.ids(SHEET) + cls.entry = kind.entries(cls.graph, cls.subjects) cls.radius = core.blast_radius(cls.graph, cls.entry) def test_shape(self): @@ -56,11 +59,11 @@ def test_aggregator_is_shared(self): self.assertEqual(len(r["affected"]), 3, sample) self.assertEqual(len(r["exclusive"]), 2, sample) (shared,) = r["shared"] - self.assertEqual(viralrecon.describe(self.graph, shared), + self.assertEqual(core_graph.describe(self.graph, shared), "MERGE_REPORT") def test_shared_reference_reaches_everything(self): - subjects = viralrecon.external_input_entry_nodes(self.graph, "ref.fa") + subjects = core_graph.external_input_entry_nodes(self.graph, "ref.fa") self.assertEqual(len(subjects["input:ref.fa"]), 2) radius = core.blast_radius(self.graph, subjects) self.assertEqual(len(radius["input:ref.fa"]["affected"]), 5) @@ -70,10 +73,10 @@ def test_missing_script_fails_closed(self): # recorded, classification must fall to IRREDUCIBLE rather than # optimistically claiming the task is regenerable. task_hash = next(iter(self.graph["tasks"])) - facts = viralrecon.classify(self.graph, task_hash, exclusive=False) + facts = contribution.classify(self.graph, task_hash, exclusive=False) self.assertEqual(facts["contribution"], "IRREDUCIBLE") # But NOT destroyed. A crate records no workdir at all, so there is - # nothing to look at — which is unverified, not gone. The previous + # nothing to look at, which is unverified, not gone. The previous # behaviour reported every crate-derived task as ALREADY_GONE, so a # whole extractor silently produced empty remediation plans. self.assertIsNone(facts["storage"]) @@ -88,7 +91,7 @@ def test_external_inputs_recognised(self): class TestRealCrateExternalForms(unittest.TestCase): """ The first real nf-prov crate recorded external inputs as https:// URLs, - relative results/ paths and #tmp entries — not the file:/// form the + relative results/ paths and #tmp entries, not the file:/// form the fixture taught this extractor to expect. All 117 were silently dropped, so a reference-update trigger found nothing and read as clean. Any object that is not another task's output must become an EXTERNAL edge. diff --git a/providers/clew-nextflow/tests/test_runs_store.py b/providers/clew-nextflow/tests/test_runs_store.py new file mode 100644 index 0000000..28c958b --- /dev/null +++ b/providers/clew-nextflow/tests/test_runs_store.py @@ -0,0 +1,41 @@ +"""--runs on a Nextflow .lineage store: names, session prefixes, resume chains.""" + +import json +import shutil +import sys +import tempfile +import unittest +from pathlib import Path + + +from clew.extract import runs + + +class LineageStoreRuns(unittest.TestCase): + """A session-id prefix names a resume chain; its newest run stands for it.""" + + def setUp(self): + from test_lineage_store import RUN_A, RUN_B, RUN_C, CHAIN, OTHER + self.root = Path(tempfile.mkdtemp()) + history = self.root / ".history" + history.mkdir() + (history / RUN_A).write_text(f"2026-08-01 10:00:00 CEST\tfirst_run\t{CHAIN}\tlid://{RUN_A}\n") + (history / RUN_B).write_text(f"2026-08-02 10:00:00 CEST\tsecond_run\t{CHAIN}\tlid://{RUN_B}\n") + (history / RUN_C).write_text(f"2026-08-03 10:00:00 CEST\tother_run\t{OTHER}\tlid://{RUN_C}\n") + self.run_b = RUN_B + + def tearDown(self): + shutil.rmtree(self.root) + + def test_a_session_prefix_resolves_to_the_chain_s_newest_run(self): + store = runs.Runs(self.root) + self.assertEqual(store.resolve("session-ch"), ("second_run", self.run_b)) + self.assertEqual(store.resolve("bbbb"), ("second_run", self.run_b)) + + def test_a_prefix_spanning_two_sessions_is_still_ambiguous(self): + with self.assertRaises(SystemExit): + runs.Runs(self.root).resolve("session-") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_sarek.py b/providers/clew-nextflow/tests/test_sarek.py similarity index 74% rename from tests/test_sarek.py rename to providers/clew-nextflow/tests/test_sarek.py index 81ed01a..7720b4a 100644 --- a/tests/test_sarek.py +++ b/providers/clew-nextflow/tests/test_sarek.py @@ -1,7 +1,7 @@ """ The sarek domain adapter: donor attribution, trigger selection, assertions. -This layer is allowed to be brittle — it parses display names — so its tests +This layer is allowed to be brittle, it parses display names, so its tests pin down the exact failure modes we have already paid for once. """ @@ -11,9 +11,13 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.domains import sarek +from clew.provider.nextflow import adapter_sarek as sarek +from clew.graph import graph as core_graph +from clew.graph import contribution +from clew.provider.nextflow import adapter as nfcore +from clew.provider.nextflow import adapter as base def graph_with_tasks(tasks): @@ -27,32 +31,32 @@ def setUp(self): self.labels = {"donor_1": "donor_1", "donor_10": "donor_10"} def test_exact_match(self): - self.assertEqual(sarek._owner_of("donor_1", self.labels), "donor_1") + self.assertEqual(base.owner_of("donor_1", self.labels), "donor_1") def test_lane_suffix_matches(self): # FASTQC reads the FASTQ directly; missing these made five tasks # invisible on the real run. The lane suffix must resolve. - self.assertEqual(sarek._owner_of("donor_1-L1", self.labels), "donor_1") + self.assertEqual(base.owner_of("donor_1-L1", self.labels), "donor_1") def test_prefix_does_not_swallow_longer_donor(self): # "donor_1" must not claim "donor_10"'s tasks. - self.assertEqual(sarek._owner_of("donor_10", self.labels), "donor_10") - self.assertEqual(sarek._owner_of("donor_10-L1", self.labels), "donor_10") + self.assertEqual(base.owner_of("donor_10", self.labels), "donor_10") + self.assertEqual(base.owner_of("donor_10-L1", self.labels), "donor_10") def test_non_donor_tags_resolve_to_nothing(self): # "(genome)" and friends use the same display-name shape. - self.assertIsNone(sarek._owner_of("genome", self.labels)) - self.assertIsNone(sarek._owner_of("genome.interval_list", self.labels)) + self.assertIsNone(base.owner_of("genome", self.labels)) + self.assertIsNone(base.owner_of("genome.interval_list", self.labels)) def test_the_longest_label_wins_whatever_the_sheet_order(self): # "KO" listed before "KO_2" used to claim "KO_2_T1". On the real # sarek run a sheet with "donor" above "donor_003" moved one FASTQC # task to the wrong donor. labels = {"KO": "KO", "KO_2": "KO_2"} - self.assertEqual(sarek._owner_of("KO_2_T1", labels), "KO_2") - self.assertEqual(sarek._owner_of("KO_T1", labels), "KO") + self.assertEqual(base.owner_of("KO_2_T1", labels), "KO_2") + self.assertEqual(base.owner_of("KO_T1", labels), "KO") labels = {"donor": "donor", "donor_003": "donor_003"} - self.assertEqual(sarek._owner_of("donor_003-L1", labels), "donor_003") + self.assertEqual(base.owner_of("donor_003-L1", labels), "donor_003") class TestSamplesheetIntegrity(unittest.TestCase): @@ -76,17 +80,17 @@ def tearDown(self): def test_a_member_under_two_subjects_is_refused_by_name(self): path = self.write([("P1", "S1"), ("P2", "S1"), ("P2", "S2")]) with self.assertRaises(SystemExit) as refused: - sarek.load_subjects(path) + sarek.Sarek().triggers['patient'].ids(path) self.assertIn("'S1' under P1, P2", str(refused.exception)) def test_a_member_that_is_another_subject_is_refused(self): path = self.write([("P1", "P2"), ("P2", "S2")]) with self.assertRaises(SystemExit): - sarek.load_subjects(path) + sarek.Sarek().triggers['patient'].ids(path) def test_a_subject_named_after_its_own_member_is_fine(self): path = self.write([("donor_001", "donor_001"), ("donor_002", "donor_002")]) - self.assertEqual(sarek.load_subjects(path), + self.assertEqual(sarek.Sarek().triggers['patient'].ids(path), {"donor_001": ["donor_001"], "donor_002": ["donor_002"]}) @@ -109,57 +113,57 @@ def setUp(self): } def test_container_selector(self): - subjects = sarek.container_entry_nodes(self.graph, "gatk4") + subjects = core_graph.container_entry_nodes(self.graph, "gatk4") self.assertEqual(subjects, {"container:gatk4": ["aa/000001"]}) def test_container_selector_empty_when_no_match(self): - subjects = sarek.container_entry_nodes(self.graph, "nonexistent") + subjects = core_graph.container_entry_nodes(self.graph, "nonexistent") self.assertEqual(subjects["container:nonexistent"], []) def test_external_input_selector_matches_basename(self): # The same reference staged at top level for one task and inside a # numbered subdirectory for another must select both. - subjects = sarek.external_input_entry_nodes(self.graph, "genome.fasta") + subjects = core_graph.external_input_entry_nodes(self.graph, "genome.fasta") self.assertEqual( subjects["input:genome.fasta"], ["aa/000001", "bb/000002"] ) def test_external_input_selector_ignores_internal_edges(self): - subjects = sarek.external_input_entry_nodes(self.graph, "out.bam") + subjects = core_graph.external_input_entry_nodes(self.graph, "out.bam") self.assertEqual(subjects["input:out.bam"], []) class TestAssertions(unittest.TestCase): def test_missing_file_means_no_assertions(self): - self.assertEqual(sarek.load_assertions(None), {}) + self.assertEqual(core_graph.load_assertions(None), {}) def test_published_assertion_sets_terminal(self): with tempfile.TemporaryDirectory() as tmp: path = Path(tmp) / "assertions.json" path.write_text(json.dumps({ - "published": [{"task": "aa/000001", "what": "Fig 3", + "released": [{"task": "aa/000001", "what": "Fig 3", "asserted_by": "someone", "date": "2026-08-21"}] })) - published = sarek.load_assertions(path) + published = core_graph.load_assertions(path) graph = graph_with_tasks({ "aa/000001": {"name": "A", "process": "A", "container": "img", "script": "run", "workdir": ""}, }) - facts = sarek.classify(graph, "aa/000001", exclusive=False, + facts = contribution.classify(graph, "aa/000001", exclusive=False, published=published) - self.assertTrue(facts["terminal"]) + self.assertTrue(facts["released"]) # The reason must carry the assertion's provenance, because the claim # is the asserter's, not Clew's. - self.assertIn("someone", facts["reason"]) + self.assertIn("someone", facts["evidence"]) def test_unpublished_task_is_not_terminal(self): graph = graph_with_tasks({ "aa/000001": {"name": "A", "process": "A", "container": "img", "script": "run", "workdir": ""}, }) - facts = sarek.classify(graph, "aa/000001", exclusive=False, published={}) - self.assertFalse(facts["terminal"]) + facts = contribution.classify(graph, "aa/000001", exclusive=False, published={}) + self.assertFalse(facts["released"]) class TestClassification(unittest.TestCase): @@ -168,7 +172,7 @@ def test_missing_script_fails_closed_to_irreducible(self): "aa/000001": {"name": "A", "process": "A", "container": "img", "script": "", "workdir": ""}, }) - facts = sarek.classify(graph, "aa/000001", exclusive=False) + facts = contribution.classify(graph, "aa/000001", exclusive=False) self.assertEqual(facts["contribution"], "IRREDUCIBLE") def test_recorded_script_and_container_is_regenerable(self): @@ -177,22 +181,22 @@ def test_recorded_script_and_container_is_regenerable(self): "aa/000001": {"name": "A", "process": "A", "container": "img", "script": "run", "workdir": tmp}, }) - facts = sarek.classify(graph, "aa/000001", exclusive=False) + facts = contribution.classify(graph, "aa/000001", exclusive=False) self.assertEqual(facts["contribution"], "REGENERABLE") def test_storage_is_unverified_until_told_where_to_look(self): # This used to assert DESTROYED, which was the bug. A path that does - # not resolve here means the artifacts were not checked — the caller - # may be on another host entirely — and reporting that as destroyed + # not resolve here means the artifacts were not checked, the caller + # may be on another host entirely, and reporting that as destroyed # produced ALREADY_GONE, i.e. "nothing to do", for work that may well # still exist and still carry an obligation. graph = graph_with_tasks({ "aa/000001": {"name": "A", "process": "A", "container": "img", "script": "run", "workdir": "/nonexistent/path"}, }) - facts = sarek.classify(graph, "aa/000001", exclusive=False) + facts = contribution.classify(graph, "aa/000001", exclusive=False) self.assertIsNone(facts["storage"]) - self.assertIn("not checked", facts["reason"]) + self.assertIn("not checked", facts["evidence"]) def test_destroyed_is_returned_only_after_actually_looking(self): graph = graph_with_tasks({ @@ -201,7 +205,7 @@ def test_destroyed_is_returned_only_after_actually_looking(self): "workdir": "/somewhere/work/aa/000001deadbeef"}, }) with tempfile.TemporaryDirectory() as work_root: - facts = sarek.classify(graph, "aa/000001", exclusive=False, + facts = contribution.classify(graph, "aa/000001", exclusive=False, work_root=work_root) self.assertEqual(facts["storage"], "DESTROYED") @@ -218,7 +222,7 @@ def test_a_graph_from_another_host_still_probes(self): "script": "run", "workdir": "/on/some/other/host/work/aa/000001deadbeef"}, }) - facts = sarek.classify(graph, "aa/000001", exclusive=False, + facts = contribution.classify(graph, "aa/000001", exclusive=False, work_root=work_root) self.assertEqual(facts["storage"], "WRITABLE") diff --git a/providers/clew-snakemake/README.md b/providers/clew-snakemake/README.md new file mode 100644 index 0000000..9b15aba --- /dev/null +++ b/providers/clew-snakemake/README.md @@ -0,0 +1,11 @@ +# clew-snakemake + +Clew provider for Snakemake: the metadata store, and a path-matching adapter. + +```bash +pip install clew-snakemake +``` + +Registers under the entry-point groups Clew discovers. `clew providers` +lists it. Built exactly as a third-party provider would be: see +[providers](../../docs/providers.md) in the engine's docs. diff --git a/providers/clew-snakemake/clew/provider/snakemake/__init__.py b/providers/clew-snakemake/clew/provider/snakemake/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/providers/clew-snakemake/clew/provider/snakemake/adapter_paths.py b/providers/clew-snakemake/clew/provider/snakemake/adapter_paths.py new file mode 100644 index 0000000..02a5872 --- /dev/null +++ b/providers/clew-snakemake/clew/provider/snakemake/adapter_paths.py @@ -0,0 +1,83 @@ +""" +A Snakemake job is named by rule and first output path, "trim (trimmed/sample_1.fq)". +PathKind finds each id in that path, bounded by separators so sample_1 stays out +of sample_10.fq. A merge named after two ids enters the graph for both. +""" + +import csv +import re + +from clew.contracts import REMOVE, Adapter, Trigger + +SEPARATORS = "-_./" +TAG_PATTERN = re.compile(r"\(([^()]+)\)\s*$") + + +def read_ids(path, column): + """{id: []} from one CSV column.""" + ids = {} + with open(path, newline="") as handle: + for row in csv.DictReader(handle): + value = (row.get(column) or "").strip() + if value: + ids.setdefault(value, []) + return ids + + +def mentions(path, value): + """Whether `value` appears in `path` bounded by separators or the ends.""" + pattern = (rf"(?:^|[{re.escape(SEPARATORS)}])" + rf"{re.escape(value)}" + rf"(?:$|[{re.escape(SEPARATORS)}])") + return re.search(pattern, path) is not None + + +class PathKind(Trigger): + """Ids from a samplesheet column, found in job output paths. A removal.""" + + mode = REMOVE + + def __init__(self, column): + self.column = column + + def add_arguments(self, parser): + parser.add_argument("--samplesheet", metavar="CSV", + help=f"ids come from its {self.column!r} column") + + def ids(self, path): + return read_ids(path, self.column) + + def entries(self, graph, ids): + # Longest id first; `mentions` stops a short id matching inside a long one. + entry = {value: set() for value in ids} + ordered = sorted(ids, key=lambda s: (-len(s), s)) + for task_hash, task in graph["tasks"].items(): + match = TAG_PATTERN.search(task.get("name", "")) + if not match: + continue + tag = match.group(1).strip() + for value in ordered: + if mentions(tag, value): + entry[value].add(task_hash) + return {value: sorted(nodes) for value, nodes in entry.items()} + + def sheet(self, args): + path = getattr(args, "samplesheet", None) + if not path: + raise SystemExit(f"--samplesheet is required: ids of this kind come from its " + f"{self.column!r} column") + return path + + def resolve(self, graph, value, args): + entry = self.entries(graph, self.ids(self.sheet(args))) + if value is not None and value not in entry: + raise SystemExit(f"unknown {self.column} {value!r}; known: {', '.join(sorted(entry))}") + return entry + + def values(self, args, graph=None): + return sorted(self.ids(self.sheet(args))) + + +class Snakemake(Adapter): + name = "snakemake" + triggers = {"sample": PathKind(column="sample")} diff --git a/clew/extract/snakemake.py b/providers/clew-snakemake/clew/provider/snakemake/extractor_metadata.py similarity index 80% rename from clew/extract/snakemake.py rename to providers/clew-snakemake/clew/provider/snakemake/extractor_metadata.py index 284e8ea..3cb8558 100644 --- a/clew/extract/snakemake.py +++ b/providers/clew-snakemake/clew/provider/snakemake/extractor_metadata.py @@ -5,7 +5,6 @@ Details and limits: docs/sources.md, "Snakemake". """ -import argparse import base64 import json import sqlite3 @@ -13,6 +12,7 @@ from pathlib import Path from clew.graph.graph import STATUS_COMPLETED, STATUS_FAILED +from clew.contracts import Extractor EXTERNAL = "EXTERNAL" METADATA_DIR = "metadata" @@ -178,12 +178,10 @@ def extract(records, workdir=""): def digests_from_consumers(edges): """ - {producer: [{file, digest}]} for every output some consumer hashed. - - The store hashes what a job read, never what it wrote, so an output's - digest is only known through its consumers. Two consumers that disagree - read different bytes under one name, a mixed store, and then the - output is left undigested rather than pinned to either. + {producer: [{file, digest}]} for every output some consumer hashed. The + store hashes what a job read, never what it wrote. Consumers that + disagree read different bytes under one name, and that output stays + undigested. """ seen = {} for edge in edges: @@ -198,33 +196,34 @@ def digests_from_consumers(edges): return details -def main(argv=None): - parser = argparse.ArgumentParser( - description="Build a Clew graph from Snakemake's metadata store.") - parser.add_argument("--workdir", required=True, - help="the workflow's working directory, or its .snakemake") - parser.add_argument("--namespace", - help="which workflow to read from a shared metadata.db") - parser.add_argument("--json-out", help="path to write the graph as JSON") - args = parser.parse_args(argv) - - records = load_records(args.workdir, args.namespace) - workdir = str(store_dir(args.workdir).parent.resolve()) - graph = extract(records, workdir) - external = [e for e in graph["edges"] if e["producer"] == EXTERNAL] - hashed = [e for e in graph["edges"] if e.get("sha256")] - incomplete = [t for t in graph["tasks"].values() if t["status"] == STATUS_FAILED] - - print(f"jobs : {len(graph['tasks'])}") - print(f" incomplete : {len(incomplete)}") - print(f"input files (edges): {len(graph['edges'])}") - print(f" external inputs : {len(external)}") - print(f" with sha256 : {len(hashed)}") - - if args.json_out: - Path(args.json_out).write_text(json.dumps(graph, indent=2)) - print(f"\nwrote {args.json_out}") - return 0 +class Snakemake(Extractor): + name = "snakemake" + description = "Snakemake's metadata store" + + def add_arguments(self, parser): + parser.add_argument("--workdir", required=True, + help="the workflow's working directory, or its .snakemake") + parser.add_argument("--namespace", + help="which workflow to read from a shared metadata.db") + + def extract(self, args): + records = load_records(args.workdir, args.namespace) + workdir = str(store_dir(args.workdir).parent.resolve()) + return extract(records, workdir) + + def summarize(self, graph, args): + external = [e for e in graph["edges"] if e["producer"] == EXTERNAL] + hashed = [e for e in graph["edges"] if e.get("sha256")] + incomplete = [t for t in graph["tasks"].values() if t["status"] == STATUS_FAILED] + print(f"jobs : {len(graph['tasks'])}") + print(f" incomplete : {len(incomplete)}") + print(f"input files (edges): {len(graph['edges'])}") + print(f" external inputs : {len(external)}") + print(f" with sha256 : {len(hashed)}") + self.coverage(graph) + + +main = Snakemake.main if __name__ == "__main__": diff --git a/providers/clew-snakemake/pyproject.toml b/providers/clew-snakemake/pyproject.toml new file mode 100644 index 0000000..9dc3282 --- /dev/null +++ b/providers/clew-snakemake/pyproject.toml @@ -0,0 +1,24 @@ +[build-system] +requires = ["setuptools>=68"] +build-backend = "setuptools.build_meta" + +[project] +name = "clew-snakemake" +version = "0.4.0" +description = "Clew provider for Snakemake: the metadata store, and a path-matching adapter" +readme = "README.md" +requires-python = ">=3.9" +license = { text = "AGPL-3.0-only" } +authors = [{ name = "QuietFlare" }] +dependencies = ["clew-lineage>=0.4"] + +[project.entry-points."clew.extractors"] +snakemake = "clew.provider.snakemake.extractor_metadata" + +[project.entry-points."clew.adapters"] +snakemake = "clew.provider.snakemake.adapter_paths" + +[tool.setuptools.packages.find] +# Installs into the clew.provider namespace; no __init__.py above the provider. +include = ["clew*"] +namespaces = true diff --git a/tests/fixtures/snakemake_diamond/metadata.db b/providers/clew-snakemake/tests/fixtures/snakemake_diamond/metadata.db similarity index 100% rename from tests/fixtures/snakemake_diamond/metadata.db rename to providers/clew-snakemake/tests/fixtures/snakemake_diamond/metadata.db diff --git a/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9yZXBvcnQudHh0 b/providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9yZXBvcnQudHh0 similarity index 100% rename from tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9yZXBvcnQudHh0 rename to providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9yZXBvcnQudHh0 diff --git a/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMS5iYW0= b/providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMS5iYW0= similarity index 100% rename from tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMS5iYW0= rename to providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMS5iYW0= diff --git a/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMS5iYW0uYmFp b/providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMS5iYW0uYmFp similarity index 100% rename from tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMS5iYW0uYmFp rename to providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMS5iYW0uYmFp diff --git a/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMV9zdGF0cw== b/providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMV9zdGF0cw== similarity index 100% rename from tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMV9zdGF0cw== rename to providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMV9zdGF0cw== diff --git a/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMi5iYW0= b/providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMi5iYW0= similarity index 100% rename from tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMi5iYW0= rename to providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMi5iYW0= diff --git a/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMi5iYW0uYmFp b/providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMi5iYW0uYmFp similarity index 100% rename from tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMi5iYW0uYmFp rename to providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMi5iYW0uYmFp diff --git a/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMl9zdGF0cw== b/providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMl9zdGF0cw== similarity index 100% rename from tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMl9zdGF0cw== rename to providers/clew-snakemake/tests/fixtures/snakemake_wildcards/metadata/cmVzdWx0cy9zMl9zdGF0cw== diff --git a/tests/test_snakemake.py b/providers/clew-snakemake/tests/test_snakemake.py similarity index 98% rename from tests/test_snakemake.py rename to providers/clew-snakemake/tests/test_snakemake.py index 3963ecb..8503808 100644 --- a/tests/test_snakemake.py +++ b/providers/clew-snakemake/tests/test_snakemake.py @@ -21,9 +21,9 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.extract import snakemake as sm +from clew.provider.snakemake import extractor_metadata as sm from clew.graph import blast_radius as core from clew.graph.graph import contract_violations, external_input_entry_nodes diff --git a/providers/clew-snakemake/tests/test_snakemake_contract.py b/providers/clew-snakemake/tests/test_snakemake_contract.py new file mode 100644 index 0000000..0bb2f7c --- /dev/null +++ b/providers/clew-snakemake/tests/test_snakemake_contract.py @@ -0,0 +1,30 @@ +"""What this extractor emits is the one graph every question reads.""" + +import unittest +from pathlib import Path + +from clew.graph.graph import STATUSES, contract_violations +from clew.provider.snakemake import extractor_metadata as sm + +FIXTURES = Path(__file__).resolve().parent / "fixtures" + + +class Contract(unittest.TestCase): + def graphs(self): + yield "files", sm.extract(sm.load_records(FIXTURES / "snakemake_wildcards")) + yield "db", sm.extract(sm.load_records(FIXTURES / "snakemake_diamond")) + + def test_conforms(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + self.assertEqual(contract_violations(graph), []) + + def test_statuses_are_in_the_vocabulary(self): + for label, graph in self.graphs(): + with self.subTest(graph=label): + for task in graph["tasks"].values(): + self.assertIn(task["status"], STATUSES) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_snakemake_domain.py b/providers/clew-snakemake/tests/test_snakemake_domain.py similarity index 84% rename from tests/test_snakemake_domain.py rename to providers/clew-snakemake/tests/test_snakemake_domain.py index e93fc3b..05f32e7 100644 --- a/tests/test_snakemake_domain.py +++ b/providers/clew-snakemake/tests/test_snakemake_domain.py @@ -10,12 +10,12 @@ import unittest from pathlib import Path -sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[3])) -from clew.domains import snakemake as dom +from clew.provider.snakemake import adapter_paths as dom from clew.graph import blast_radius as core -ROOT = Path(__file__).resolve().parent.parent +ROOT = Path(__file__).resolve().parents[3] SAMPLES = ("sample_1", "sample_2", "sample_3") @@ -65,7 +65,7 @@ def setUp(self): self.graph = snk_run_graph() def test_a_sample_enters_at_its_own_trim_align_and_stats(self): - entry = dom.subject_entry_nodes(self.graph, {s: [] for s in SAMPLES}) + entry = dom.Snakemake().triggers['sample'].entries(self.graph, {s: [] for s in SAMPLES}) self.assertEqual(entry["sample_1"], [ "align/aligned/sample_1.bam", "stats/stats/sample_1.stats", "trim/trimmed/sample_1.fq"]) @@ -74,12 +74,12 @@ def test_a_sample_enters_at_its_own_trim_align_and_stats(self): def test_sample_1_does_not_swallow_sample_10(self): node, task = job("trim", "trimmed/sample_10.fq") self.graph["tasks"][node] = task - entry = dom.subject_entry_nodes(self.graph, {"sample_1": [], "sample_10": []}) + entry = dom.Snakemake().triggers['sample'].entries(self.graph, {"sample_1": [], "sample_10": []}) self.assertNotIn(node, entry["sample_1"]) self.assertEqual(entry["sample_10"], [node]) def test_multiqc_is_reached_as_shared(self): - entry = dom.subject_entry_nodes(self.graph, {s: [] for s in SAMPLES}) + entry = dom.Snakemake().triggers['sample'].entries(self.graph, {s: [] for s in SAMPLES}) radius = core.blast_radius(self.graph, entry)["sample_1"] self.assertEqual(radius["shared"], {"multiqc/report/multiqc.txt"}) self.assertEqual(len(radius["exclusive"]), 3) @@ -90,7 +90,7 @@ def test_the_sample_column_is_read_by_default(self): with tempfile.TemporaryDirectory() as tmp: sheet = Path(tmp, "samplesheet.csv") sheet.write_text("sample,fastq_1\nsample_1,raw/sample_1.fq\nsample_9,raw/sample_9.fq\n") - self.assertEqual(dom.load_subjects(sheet), {"sample_1": [], "sample_9": []}) + self.assertEqual(dom.Snakemake().triggers['sample'].ids(sheet), {"sample_1": [], "sample_9": []}) class Cli(unittest.TestCase): @@ -103,14 +103,14 @@ def test_impact_takes_the_snakemake_pipeline(self): result = subprocess.run( [sys.executable, "-m", "clew.questions.impact", "--graph", str(graph), "--pipeline", "snakemake", "--samplesheet", str(sheet), - "--subject", "sample_1", "--json", "-"], + "--trigger", "sample:sample_1", "--json", "-"], capture_output=True, text=True, cwd=ROOT) self.assertEqual(result.returncode, 0, result.stderr) plan = json.loads(result.stdout[result.stdout.index("{"):]) self.assertEqual(plan["tasks_affected"], 4) by_task = {i["task"]: i for i in plan["plan"]} - self.assertFalse(by_task["multiqc/report/multiqc.txt"]["exclusive"]) - self.assertTrue(by_task["trim/trimmed/sample_1.fq"]["exclusive"]) + self.assertEqual(by_task["multiqc/report/multiqc.txt"]["scope"], "shared") + self.assertEqual(by_task["trim/trimmed/sample_1.fq"]["scope"], "exclusive") if __name__ == "__main__": diff --git a/pyproject.toml b/pyproject.toml index fd908be..0095ce5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -30,6 +30,14 @@ dependencies = [] [project.optional-dependencies] # Only the event log needs a driver. Everything else stays stdlib. log = ["psycopg[binary]>=3.1"] +# The providers Clew ships, each its own distribution under providers/. +nextflow = ["clew-nextflow"] +snakemake = ["clew-snakemake"] +cromwell = ["clew-cromwell"] +horus = ["clew-horus"] +dnanexus = ["clew-dnanexus"] +latch = ["clew-latch"] +all = ["clew-nextflow", "clew-snakemake", "clew-cromwell", "clew-horus", "clew-dnanexus", "clew-latch"] [project.urls] Homepage = "https://quietflare.net/clew" @@ -39,8 +47,11 @@ Issues = "https://github.com/QuietFlare/clew/issues" [project.scripts] clew = "clew.__main__:main" + [tool.setuptools.packages.find] +# clew is a namespace: provider packages install into clew.provider.* include = ["clew*"] +namespaces = true [tool.setuptools.package-data] clew = ["data/*.json", "data/*.csv", "data/samplesheets/*.csv"] diff --git a/tests/test_adapter_registry.py b/tests/test_adapter_registry.py new file mode 100644 index 0000000..8376420 --- /dev/null +++ b/tests/test_adapter_registry.py @@ -0,0 +1,47 @@ +""" +The domain contract: defining a named subclass is the registration. +""" + +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from clew.contracts import Adapter, discover + + +class TestRegistry(unittest.TestCase): + def test_builtins_register_by_import(self): + names = set(discover(Adapter)) + self.assertTrue({"sarek", "rnaseq", "viralrecon", "snakemake"} <= names) + + def test_a_named_subclass_registers_itself(self): + class Custom(Adapter): + name = "custom-test" # no kinds of its own; the engine's still apply + + try: + self.assertIsInstance(discover(Adapter)["custom-test"], Custom) + finally: + Adapter.registered.pop("custom-test", None) + + def test_a_nameless_subclass_is_a_base_not_a_provider(self): + class Shared(Adapter): + pass + + self.assertNotIn(None, Adapter.registered) + self.assertFalse(any(isinstance(d, Shared) for d in Adapter.registered.values())) + + def test_engine_kinds_resolve_for_a_domain_with_none(self): + from clew.contracts.trigger import lookup + class Bare(Adapter): + name = "bare-test" + try: + self.assertIsNotNone(lookup(Bare(), "container")) + self.assertIsNone(lookup(Bare(), "patient", {"tasks": {}, "edges": []})) + finally: + Adapter.registered.pop("bare-test", None) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_blast_radius.py b/tests/test_blast_radius.py index e80378d..d623bf3 100644 --- a/tests/test_blast_radius.py +++ b/tests/test_blast_radius.py @@ -3,7 +3,7 @@ CLAUDE.md: mixed verdicts from one node, and the load-bearing input. Synthetic graphs use readable ids ("pool", "paper") precisely because core -must not care — if these tests pass, core never looked inside the strings. +must not care, if these tests pass, core never looked inside the strings. """ import sys @@ -116,7 +116,7 @@ class TestMixedVerdictsFromOneNode(unittest.TestCase): """ The withdrawn donor fed a pool. The pool fed BOTH a published paper and an unpublished analysis. One traversal must produce two different - answers — NOTIFY_ONLY on the published branch, REGENERATE on the other. + answers, NOTIFY_ONLY on the published branch, REGENERATE on the other. If the model cannot do this, it is wrong. """ @@ -138,8 +138,8 @@ def test_two_answers_from_one_pool(self): for node in affected: verdicts[node] = policy.remediate( c.REGENERABLE, - exclusive=node in exclusive, - terminal=node in published, + scope=policy.EXCLUSIVE if node in exclusive else policy.SHARED, + released=node in published, ) self.assertEqual(verdicts["paper"], c.NOTIFY_ONLY) @@ -152,7 +152,7 @@ def test_two_answers_from_one_pool(self): class TestLoadBearingInput(unittest.TestCase): """ Removing a reference/control invalidates everything calibrated against - it — artifacts that are NOT downstream of any donor. The trigger is the + it, artifacts that are NOT downstream of any donor. The trigger is the input itself, and the blast radius must reach past the donors entirely. """ @@ -174,8 +174,7 @@ def test_reference_reaches_what_donors_do_not(self): # But not things that never touched it. self.assertNotIn("unrelated", affected) - # A donor's own radius does NOT cover the other donor's artifacts — - # only the reference trigger reaches both. That asymmetry is the point. + # A donor's own radius does NOT cover the other donor's artifacts, # only the reference trigger reaches both. That asymmetry is the point. donor_radius = core.blast_radius(g, {"d1": ["d1"]}) self.assertNotIn("align2", donor_radius["d1"]["affected"]) diff --git a/tests/test_class_hook.py b/tests/test_class_hook.py new file mode 100644 index 0000000..cc63e1d --- /dev/null +++ b/tests/test_class_hook.py @@ -0,0 +1,66 @@ +"""An adapter that knows a step may say its class; the plan records that it did.""" + +import io +import json +import sys +import unittest +from contextlib import redirect_stdout, redirect_stderr +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from clew.contracts import Adapter +from clew.provider.nextflow import adapter_sarek as sarek +from clew.questions import impact + +GRAPH = str(Path(__file__).resolve().parent.parent / "clew" / "data" / "graph5.json") +MULTIQC = "c9/023b13" + + +class Knows(sarek.Sarek): + name = "sarek-test-classes" + answer = None + + def contribution(self, graph, task_hash, kind): + type(self).asked_kind = kind + return self.answer if task_hash == MULTIQC else None + + +def plan_items(*argv): + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + impact.main(["--pipeline", "sarek-test-classes", "--graph", GRAPH, + "--container", "gatk4", "--json", "-", *argv]) + text = out.getvalue() + return {i["task"]: i for i in json.loads(text[text.index("{"):])["plan"]} + + +class TestClassHook(unittest.TestCase): + @classmethod + def tearDownClass(cls): + Adapter.registered.pop("sarek-test-classes", None) + + def test_adapter_answer_replaces_evidence_and_is_recorded(self): + Knows.answer = "SEPARABLE" + item = plan_items()[MULTIQC] + self.assertEqual(item["contribution"], "SEPARABLE") + self.assertEqual(item["class_asserted_by"], "sarek-test-classes") + self.assertIn("evidence alone said REGENERABLE", item["evidence"]) + self.assertEqual(Knows.asked_kind, "container") + + def test_none_keeps_the_evidence_answer(self): + Knows.answer = None + item = plan_items()[MULTIQC] + self.assertEqual(item["contribution"], "REGENERABLE") + self.assertNotIn("class_asserted_by", item) + + def test_an_unknown_class_is_refused_by_name(self): + Knows.answer = "REMOVABLE" + with self.assertRaises(SystemExit) as stop: + plan_items() + self.assertIn("REMOVABLE", str(stop.exception)) + self.assertIn("SEPARABLE, REGENERABLE, IRREDUCIBLE", str(stop.exception)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_contribution.py b/tests/test_contribution.py index a54eaac..9b23122 100644 --- a/tests/test_contribution.py +++ b/tests/test_contribution.py @@ -1,7 +1,7 @@ """ The vocabulary: the classes, the actions, and the fail-closed normalisation. -Deciding is not tested here — it moved to core/policy.py, where the table has +Deciding is not tested here, it moved to core/policy.py, where the table has a version and a hash. See tests/test_policy.py. The split is deliberate: the words have to be stable for Clew to mean anything, while the table has to be versioned so a plan from March can be replayed under March's table. @@ -65,18 +65,18 @@ def classify(self, **task): def test_missing_container_is_named(self): facts = self.classify(script="run", container="") self.assertEqual(facts["contribution"], c.IRREDUCIBLE) - self.assertIn("no container recorded", facts["reason"]) + self.assertIn("no container recorded", facts["evidence"]) def test_missing_script_is_named(self): facts = self.classify(script="", container="img") self.assertEqual(facts["contribution"], c.IRREDUCIBLE) - self.assertIn("no script recorded", facts["reason"]) + self.assertIn("no script recorded", facts["evidence"]) def test_both_missing_names_both(self): facts = self.classify(script="", container="") - self.assertIn("no script or container recorded", facts["reason"]) + self.assertIn("no script or container recorded", facts["evidence"]) def test_both_recorded_is_regenerable(self): facts = self.classify(script="run", container="img") self.assertEqual(facts["contribution"], c.REGENERABLE) - self.assertIn("script and container recorded", facts["reason"]) + self.assertIn("script and container recorded", facts["evidence"]) diff --git a/tests/test_core_boundary.py b/tests/test_core_boundary.py index 8455ba8..02ac1eb 100644 --- a/tests/test_core_boundary.py +++ b/tests/test_core_boundary.py @@ -23,7 +23,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) PACKAGE = Path(__file__).resolve().parent.parent / "clew" -CLEAN = ["graph", "ledger"] +CLEAN = ["graph", "ledger", "contracts"] # Command modules talk to people and may use their words. Only the library # modules under a clean package are held to the rule. COMMANDS = {"evidence.py", "logbook.py", "rulebook.py"} @@ -36,11 +36,11 @@ ALLOWED = { "graph": set(), - "domains": {"graph"}, - "ledger": {"graph"}, - "extract": {"graph", "domains"}, + "contracts": {"graph"}, + "ledger": {"graph", "contracts"}, + "extract": {"graph", "contracts"}, "views": {"graph", "ledger"}, - "questions": {"graph", "domains", "ledger", "extract", "views"}, + "questions": {"graph", "ledger", "extract", "views", "contracts"}, } IMPORT = re.compile(r"^\s*(?:from|import)\s+clew\.(\w+)", re.MULTILINE) @@ -75,5 +75,81 @@ def test_packages_import_downward_only(self): self.assertEqual(offences, [], "\n".join(offences)) +PROVIDERS = Path(__file__).resolve().parent.parent / "providers" +# What a provider may import from clew: the graph, the contracts, and the +# engine-side extract tools. Not another provider, not the questions. +PROVIDER_ALLOWED = {"graph", "contracts", "extract"} +PROVIDER_IMPORT = re.compile(r"^\s*(?:from|import)\s+clew\.provider\.(\w+)", re.MULTILINE) + + +def declared_entry_points(pyproject, group): + """[(name, module)] under one entry-point group. A regex, since tomllib is 3.11+.""" + text = pyproject.read_text() + section = re.search(rf'^\[project\.entry-points\."{re.escape(group)}"\]\n(.*?)(?=^\[|\Z)', + text, re.M | re.S) + if not section: + return [] + return re.findall(r'^([\w-]+)\s*=\s*"([^"]+)"', section.group(1), re.M) + + +class TestProviders(unittest.TestCase): + """The built-ins are held to what a third-party provider could do.""" + + def test_providers_reach_only_the_public_surface(self): + offences = [] + for package in sorted(PROVIDERS.glob("*/clew/provider/*")): + for path in sorted(package.glob("*.py")): + text = path.read_text() + for target in IMPORT.findall(text): + if target != "provider" and target not in PROVIDER_ALLOWED: + offences.append(f"{package.name}/{path.name} imports clew.{target}") + for other in PROVIDER_IMPORT.findall(text): + if other != package.name: + offences.append(f"{package.name}/{path.name} imports provider {other}") + self.assertEqual(offences, [], "\n".join(offences)) + + def test_namespace_levels_carry_no_init(self): + # clew and clew.provider are namespace packages. An __init__.py at + # either level, in any distribution, claims the whole package for that + # one directory and every other provider silently stops registering. + offences = [str(p.relative_to(PACKAGE.parent)) + for p in (PACKAGE / "__init__.py", PACKAGE / "provider" / "__init__.py") + if p.exists()] + for dist in sorted(PROVIDERS.glob("*")): + for level in (dist / "clew" / "__init__.py", + dist / "clew" / "provider" / "__init__.py"): + if level.exists(): + offences.append(str(level.relative_to(PACKAGE.parent))) + self.assertEqual(offences, [], "namespace level has an __init__.py:\n " + + "\n ".join(offences)) + + def test_every_provider_is_discoverable(self): + # Each provider directory must be reachable through the namespace, and + # every entry point its pyproject declares must import and register. + import importlib + from clew.contracts import Adapter, Extractor + groups = {"clew.adapters": Adapter, "clew.extractors": Extractor} + offences = [] + for dist in sorted(PROVIDERS.glob("*")): + name = next((dist / "clew" / "provider").glob("*")).name + try: + importlib.import_module(f"clew.provider.{name}") + except ImportError as exc: + offences.append(f"{dist.name}: clew.provider.{name} not importable: {exc}") + continue + for group, contract in groups.items(): + for key, module in declared_entry_points(dist / "pyproject.toml", group): + importlib.import_module(module) + if key not in contract.registered: + offences.append(f"{dist.name}: {group} entry {key!r} imported " + f"{module} but nothing registered under that name") + self.assertEqual(offences, [], "\n".join(offences)) + + def test_no_provider_code_remains_in_the_engine(self): + self.assertFalse((PACKAGE / "domains").exists()) + self.assertEqual(sorted(p.name for p in (PACKAGE / "extract").glob("*.py")), + ["__init__.py", "digest.py", "runs.py", "stitch.py"]) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_dashboard.py b/tests/test_dashboard.py index 9f6a655..48183ee 100644 --- a/tests/test_dashboard.py +++ b/tests/test_dashboard.py @@ -14,7 +14,7 @@ It must ESCAPE EVERYTHING. Subjects, actors, triggers and event types are strings the recording organisation chose, and they go straight into markup. An identifier containing a tag would otherwise break the page at best and -inject script at worst — in a document whose entire purpose is being trusted. +inject script at worst, in a document whose entire purpose is being trusted. """ import json @@ -48,7 +48,7 @@ def log_entry(seq, prev_hash, subject="s1", event_type="Withdrawn", return fields -def a_plan(trigger="withdrawal of s1", undetermined=False): +def a_plan(trigger="removal of s1", undetermined=False): if undetermined: decision = policy_module.decide("REGENERABLE", storage=None) storage = None @@ -56,16 +56,16 @@ def a_plan(trigger="withdrawal of s1", undetermined=False): decision = policy_module.decide("REGENERABLE", storage="WRITABLE") storage = "WRITABLE" return { - "clew_plan_version": 1, "trigger": trigger, + "clew_plan_version": 2, "trigger": trigger, **policy_module.identify(policy_module.DEFAULT), "tasks_total": 10, "tasks_affected": 1, "entry_tasks": ["t0"], "actions": {decision["action"] or policy_module.UNDETERMINED: 1}, "plan": [{ "task": "t1", "process": "P", "name": "t1", "action": decision["action"], "rule": decision["rule"], - "because": decision["because"], "possible": decision.get("possible"), + "reason": decision["reason"], "possible": decision.get("possible"), "contribution": "REGENERABLE", "storage": storage, - "exclusive": False, "terminal": False, "reason": "test", + "scope": "shared", "released": False, "evidence": "test", "evidence_path": ["t0", "t1"], }], "caveats": ["a stated limit"], diff --git a/tests/test_drift.py b/tests/test_drift.py index db2f311..ce06734 100644 --- a/tests/test_drift.py +++ b/tests/test_drift.py @@ -182,31 +182,5 @@ def test_an_input_without_a_digest_cannot_key_a_pair(self): self.assertEqual(verdict["a1"], drift.UNVERIFIED) -class SameChain(unittest.TestCase): - """ - The store holds one graph per resume chain. Two runs of one chain - load the same graph, and drift compared it with itself. - """ - - def test_two_runs_of_one_session_are_refused(self): - import contextlib - import io - import tempfile - from tests.test_lineage_store import RUN_A, RUN_B, CHAIN, PRODUCER, task_run, write_record - with tempfile.TemporaryDirectory() as tmp: - store = Path(tmp) - (store / ".history").mkdir() - (store / ".history" / RUN_A).write_text( - f"2026-08-01 10:00:00 CEST\tfirst_run\t{CHAIN}\tlid://{RUN_A}\n") - (store / ".history" / RUN_B).write_text( - f"2026-08-02 10:00:00 CEST\tsecond_run\t{CHAIN}\tlid://{RUN_B}\n") - write_record(store / PRODUCER, task_run(CHAIN, RUN_A, "PIPE:ALIGN")) - with self.assertRaises(SystemExit) as refused, \ - contextlib.redirect_stdout(io.StringIO()): - drift.main(["--runs", tmp, "--before", "first_run", "--after", "second_run"]) - self.assertIn("same graph", str(refused.exception)) - self.assertIn("resume chain", str(refused.exception)) - - if __name__ == "__main__": unittest.main() diff --git a/tests/test_eventlog.py b/tests/test_eventlog.py index 7d6d4a4..921f0f6 100644 --- a/tests/test_eventlog.py +++ b/tests/test_eventlog.py @@ -7,7 +7,7 @@ required installing a database driver and standing up a server, "anyone can check this without us" would be a slogan rather than a fact. -Storage behaviour — the role grants, the triggers, concurrent appends — lives +Storage behaviour, the role grants, the triggers, concurrent appends, lives in test_eventlog_postgres.py and needs a server. """ diff --git a/tests/test_eventlog_postgres.py b/tests/test_eventlog_postgres.py index b0d8c01..17e7eca 100644 --- a/tests/test_eventlog_postgres.py +++ b/tests/test_eventlog_postgres.py @@ -6,7 +6,7 @@ 1. GRANTS stop the application. The writer role holds SELECT and INSERT and was never granted UPDATE, DELETE or TRUNCATE. The tests below check that - the writer is refused with a PRIVILEGE error, not a trigger error — if + the writer is refused with a PRIVILEGE error, not a trigger error, if the trigger fired first, the grants would be untested and a future "let's simplify the DDL" would remove the real protection unnoticed. @@ -63,7 +63,7 @@ class LogTestCase(unittest.TestCase): # Roles are cluster-global, not per-database. Using the real role names # here would mean running the test suite RESETS the passwords of the - # production writer and auditor — on any cluster that happens to host + # production writer and auditor, on any cluster that happens to host # both. Test roles get test names. WRITER_ROLE = "clew_test_writer" AUDITOR_ROLE = "clew_test_auditor" @@ -124,8 +124,7 @@ def test_writer_is_refused_truncate_by_privilege(self): cur.execute("TRUNCATE events") def test_writer_cannot_grant_itself_more(self): - # Postgres does not ERROR on a grant by a role without grant option — - # it warns and does nothing. So assert the outcome, not the exception: + # Postgres does not ERROR on a grant by a role without grant option, # it warns and does nothing. So assert the outcome, not the exception: # after trying, the writer still cannot update. Testing for a raised # error here would have passed for the wrong reason on some versions # and silently stopped testing anything on others. @@ -259,7 +258,7 @@ def test_an_unparseable_effective_from_is_refused(self): def test_timestamps_survive_the_round_trip_byte_for_byte(self): # Why the columns are text. A timestamptz would come back in the # server's own formatting and every later hash would fail to - # recompute — verification broken by a display convention. + # recompute, verification broken by a display convention. odd = "2026-01-01T00:00:00+00:00" el.append(self.writer, "T", "s", actor="t", effective_from=odd) self.assertEqual(el.read(self.writer)[0]["effective_from"], odd) @@ -305,7 +304,7 @@ def test_missing_anchor_raises_rather_than_falling_back(self): class TestConcurrentAppend(LogTestCase): """ The reason for the advisory lock. Without it two appenders read the same - head and the chain forks — which the PRIMARY KEY would turn into a crash, + head and the chain forks, which the PRIMARY KEY would turn into a crash, and a weaker schema would turn into silent corruption. """ diff --git a/tests/test_evidence.py b/tests/test_evidence.py index 9f5e259..08a74d2 100644 --- a/tests/test_evidence.py +++ b/tests/test_evidence.py @@ -3,7 +3,7 @@ The tests that matter here are the forgeries. A bundle that verifies when nothing is wrong is unremarkable. What has to hold is that a bundle someone -has quietly improved cannot pass — including the careful forgery, where the +has quietly improved cannot pass, including the careful forgery, where the manifest is rebuilt so every hash matches and only the conclusion changed. That one is caught by replay, which is the check most evidence packages do not have. @@ -55,25 +55,24 @@ def a_plan(policy_document=None): ("t2", "REGENERABLE", "DESTROYED", False, False), ("t3", "REGENERABLE", "WRITABLE", True, False), ("t4", "IRREDUCIBLE", "WRITABLE", False, True), - # Published AND destroyed: the one combination v1 and v2 disagree on, - # so a plan without it would replay happily under either table and - # the wrong-policy check would pass for the wrong reason. + # Released AND destroyed: release must win, and a plan without the + # combination would not notice a table that asks existence first. ("t5", "REGENERABLE", "DESTROYED", False, True), ] items = [] for task, klass, storage, exclusive, terminal in facts: decision = policy_module.decide(klass, storage=storage, - exclusive=exclusive, terminal=terminal, + scope=(policy_module.EXCLUSIVE if exclusive else policy_module.SHARED), released=terminal, policy=policy_document) items.append({ "task": task, "process": "P", "name": task, "action": decision["action"], "rule": decision["rule"], - "because": decision["because"], "contribution": klass, - "storage": storage, "exclusive": exclusive, "terminal": terminal, - "reason": "test", + "reason": decision["reason"], "contribution": klass, + "storage": storage, "scope": (policy_module.EXCLUSIVE if exclusive else policy_module.SHARED), "released": terminal, + "evidence": "test", }) return { - "clew_plan_version": 1, + "clew_plan_version": 2, **policy_module.identify(policy_document), "trigger": "test:trigger", "tasks_total": 10, @@ -124,10 +123,10 @@ def setUp(self): self.addCleanup(holder.cleanup) self.tmp = holder.name - def run_cli(self, *args): + def run_cli(self, *args, cwd=None): return subprocess.run( [sys.executable, "-m", "clew.ledger.evidence", *args], - capture_output=True, text=True) + capture_output=True, text=True, cwd=cwd) class TestSealing(BundleTestCase): @@ -254,7 +253,7 @@ class TestWitness(BundleTestCase): The check that closes the log's open gap. A truncated chain is internally consistent, so verify() on the log alone - passes — nothing inside a database can notice something that is no longer + passes, nothing inside a database can notice something that is no longer in it. A bundle notices, because it left the building carrying the head it saw. """ @@ -314,7 +313,8 @@ def test_matching_policy_passes(self): def test_a_swapped_policy_is_caught(self): # The plan and the table it was decided under cannot drift apart. - plan = a_plan(policy_module.V2) + other = policy_module.validate(dict(policy_module.V1, version="other")) + plan = a_plan(other) check = bundle.verify_policy(plan, policy_module.V1) self.assertFalse(check["ok"]) self.assertIn("hashes to", check["detail"]) @@ -350,11 +350,13 @@ def test_a_decorative_rule_citation_is_caught(self): self.assertIn("rule", check["detail"]) def test_replaying_under_the_wrong_policy_is_caught(self): - # A v1 plan does not reproduce under v2. Bundling today's table with - # yesterday's plan would produce a bundle that passes for the wrong - # reason, which is worse than no bundle. + # Bundling a stricter table with a plan decided under v1 would give a + # bundle that passes for the wrong reason, which is worse than none. + strict = policy_module.validate({"version": "strict", "rules": [ + policy_module.rule("S1", "QUARANTINE", "never regenerate", + contribution="REGENERABLE")] + policy_module.V1["rules"]}) plan = a_plan(policy_module.V1) - self.assertFalse(bundle.verify_replay(plan, policy_module.V2)["ok"]) + self.assertFalse(bundle.verify_replay(plan, strict)["ok"]) class TestEndToEnd(BundleTestCase): @@ -374,6 +376,20 @@ def test_build_then_verify(self): for check in ("files", "log", "policy", "replay"): self.assertIn(check, checked.stdout) + def test_without_out_the_bundle_is_named_after_the_trigger(self): + plan = a_plan() + plan["trigger"] = "container build:1.2.3" + plan_path = Path(self.tmp) / "plan.json" + plan_path.write_text(json.dumps(plan)) + + built = self.run_cli("build", "--plan", str(plan_path), cwd=self.tmp) + self.assertEqual(built.returncode, 0, built.stderr) + + made = [p for p in Path(self.tmp).iterdir() if p.is_dir()] + self.assertEqual(len(made), 1) + self.assertRegex(made[0].name, r"^container-build-1\.2\.3-\d{4}-\d{2}-\d{2}$") + self.assertEqual(self.run_cli("verify", str(made[0])).returncode, 0) + def test_a_bundle_with_no_log_says_so_in_its_coverage(self): # Never claim completeness. A bundle anchored to no log head cannot # detect a later truncation of anything, and must say that. @@ -411,8 +427,10 @@ def test_sealing_a_plan_under_a_policy_it_did_not_use_is_refused(self): plan = a_plan(policy_module.V1) plan_path = Path(self.tmp) / "plan.json" plan_path.write_text(json.dumps(plan)) + other = Path(self.tmp) / "other.json" + other.write_text(json.dumps(dict(policy_module.V1, version="other"))) built = self.run_cli("build", "--out", str(Path(self.tmp) / "b"), - "--plan", str(plan_path), "--policy", "v2") + "--plan", str(plan_path), "--policy", str(other)) self.assertNotEqual(built.returncode, 0) self.assertIn("policy mismatch", built.stderr) @@ -559,14 +577,14 @@ def test_a_narrowed_possible_map_is_caught(self): # An undetermined item's candidates are its whole content. "One of # three" quietly becoming "one of one" reads as settled. decision = policy_module.decide("REGENERABLE", storage=None, - exclusive=True) + scope=policy_module.EXCLUSIVE) plan = a_plan() plan["plan"].append({ "task": "t6", "process": "P", "name": "t6", "action": None, - "rule": None, "because": decision["because"], + "rule": None, "reason": decision["reason"], "possible": {"ALREADY_GONE": "R1"}, "contribution": "REGENERABLE", "storage": None, - "exclusive": True, "terminal": False, "reason": "test"}) + "scope": "exclusive", "released": False, "evidence": "test"}) plan["tasks_affected"] = 6 plan["actions"] = action_counts(plan["plan"]) check = bundle.verify_replay(plan, policy_module.DEFAULT) @@ -576,7 +594,7 @@ def test_a_narrowed_possible_map_is_caught(self): def test_a_null_terminal_cannot_replay_to_a_settled_verdict(self): plan = a_plan() - plan["plan"][0].update(terminal=None, storage=None, + plan["plan"][0].update(released=None, storage=None, action="NOTIFY_ONLY", rule="R2") check = bundle.verify_replay(plan, policy_module.DEFAULT) self.assertFalse(check["ok"]) @@ -698,24 +716,3 @@ def test_since_without_previous_is_refused(self): if __name__ == "__main__": unittest.main() - - -class TestPreviousBundle(BundleTestCase): - """--previous must name a sealed bundle, and say so when it does not.""" - - def run_cli(self, *args): - return subprocess.run( - [sys.executable, "-m", "clew.ledger.evidence", *args], - capture_output=True, text=True) - - def test_a_directory_with_no_manifest_is_refused_with_a_message(self): - plan_path = Path(self.tmp) / "plan.json" - plan_path.write_text(json.dumps(a_plan())) - empty = Path(self.tmp) / "not-a-bundle" - empty.mkdir() - result = self.run_cli("build", "--out", str(Path(self.tmp) / "b"), - "--plan", str(plan_path), - "--previous", str(empty)) - self.assertNotEqual(result.returncode, 0) - self.assertNotIn("Traceback", result.stderr) - self.assertIn("not a sealed bundle", result.stderr) diff --git a/tests/test_extractor_registry.py b/tests/test_extractor_registry.py new file mode 100644 index 0000000..77461b3 --- /dev/null +++ b/tests/test_extractor_registry.py @@ -0,0 +1,73 @@ +""" +The extractor contract: defining a named subclass is the registration, and +the base checks every graph against the schema before writing it. +""" + +import io +import sys +import tempfile +import unittest +from contextlib import redirect_stdout +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from clew.contracts import Extractor, discover + + +def good_graph(): + return {"tasks": {"a/1": {"hash": "a/1", "name": "x", "process": "x", + "container": "", "status": "COMPLETED", + "script": "", "workdir": "/w/a/1"}}, + "edges": [], "outputs": {}} + + +class Fake(Extractor): + name = "fake-test" + description = "a test double" + graph = None + + def add_arguments(self, parser): + parser.add_argument("--source", required=True) + + def extract(self, args): + return self.graph + + +class TestRegistry(unittest.TestCase): + @classmethod + def tearDownClass(cls): + Extractor.registered.pop("fake-test", None) + + def test_builtins_register_by_import(self): + names = set(discover(Extractor)) + self.assertTrue({"nextflow", "nextflow-work", "ro-crate", "horus", "cromwell", + "snakemake", "dnanexus", "latch"} <= names) + + def test_a_named_subclass_registers_itself(self): + self.assertIsInstance(discover(Extractor)["fake-test"], Fake) + + def test_a_provider_missing_a_method_fails_at_definition(self): + with self.assertRaises(TypeError): + class Bare(Extractor): + name = "bare-test" + + def test_run_writes_a_conforming_graph(self): + Fake.graph = good_graph() + with tempfile.TemporaryDirectory() as tmp: + out = Path(tmp) / "g.json" + with redirect_stdout(io.StringIO()) as printed: + code = Fake.main(["--source", "x", "--json-out", str(out)]) + self.assertEqual(code, 0) + self.assertTrue(out.exists()) + self.assertIn("tasks : 1", printed.getvalue()) + + def test_run_refuses_a_graph_that_breaks_the_contract(self): + Fake.graph = {"tasks": {"a/1": {"hash": "a/1"}}, "edges": [], "outputs": {}} + with self.assertRaises(SystemExit) as stop: + Fake.main(["--source", "x"]) + self.assertIn("breaks the contract", str(stop.exception)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_gate.py b/tests/test_gate.py index 2a71aaf..add1412 100644 --- a/tests/test_gate.py +++ b/tests/test_gate.py @@ -117,7 +117,7 @@ def test_log_order_breaks_a_same_day_tie(self): class TestAsOf(unittest.TestCase): """ - 'Was this run permitted when we ran it?' — a different question from + 'Was this run permitted when we ran it?', a different question from 'is it permitted now', and both have to be answerable. """ @@ -219,8 +219,8 @@ def test_an_unreachable_log_stops_the_build(self): # The one that matters most. An unreachable log is not an absence of # prohibitions, and a green build here would be a lie. # - # Two ways of not reaching it — no driver installed, or nothing - # listening — and the assertion is on the property both must have + # Two ways of not reaching it, no driver installed, or nothing + # listening, and the assertion is on the property both must have # rather than on either message. Pinning one string would have let # the other path regress to exit 0 unnoticed. result = self.run_gate( @@ -269,7 +269,7 @@ def test_a_missing_samplesheet_is_refused_with_a_message(self): "--dsn", "postgresql://nowhere/none") self.assertNotEqual(result.returncode, 0) self.assertNotIn("Traceback", result.stderr) - self.assertIn("cannot read --samplesheet", result.stderr) + self.assertIn("cannot read the patient inputs", result.stderr) def test_the_shipped_policy_template_is_valid_json_and_names_types(self): template = json.loads((ROOT / "clew" / "data" / "gate-policy.example.json").read_text()) diff --git a/tests/test_graph_contract.py b/tests/test_graph_contract.py index 0572202..0e7c1a4 100644 --- a/tests/test_graph_contract.py +++ b/tests/test_graph_contract.py @@ -1,7 +1,7 @@ """ Every extractor emits the same graph, and everything downstream assumes -its shape. This runs the contract over every shipped graph and every -fixture-built one, so a new extractor joins by adding one line. +its shape. The shipped graphs are checked here; each provider checks its +own extractor's output in its own tests. """ import json @@ -12,34 +12,16 @@ sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) -import clew -from clew.extract import cromwell as cw -from clew.extract import dnanexus as dx -from clew.extract import horus as hz -from clew.extract import latch as lt -from clew.extract import rocrate as rc -from clew.extract import snakemake as sm +from importlib.metadata import version from clew.graph.graph import STATUSES, contract_violations, task_status ROOT = Path(__file__).resolve().parent.parent DATA = ROOT / "clew" / "data" -FIXTURES = Path(__file__).resolve().parent / "fixtures" SHIPPED = ["graph5.json", "graph_rna.json", "graph_vr.json", "graph_da.json", "graph_chain.json"] -def fixture_graphs(): - yield "horus", hz.extract(FIXTURES / "horus_run") - yield "rocrate", rc.extract(FIXTURES / "ro-crate-metadata.json") - yield "dnanexus", dx.extract(dx.load_records(FIXTURES / "dnanexus")) - yield "latch", lt.extract(lt.load_records(FIXTURES / "latch")) - yield "cromwell", cw.extract(cw.load_metadata(FIXTURES / "cromwell" / "diamond.json")) - yield "cromwell-sub", cw.extract(cw.load_metadata(FIXTURES / "cromwell" / "outer.json")) - yield "snakemake-files", sm.extract(sm.load_records(FIXTURES / "snakemake_wildcards")) - yield "snakemake-db", sm.extract(sm.load_records(FIXTURES / "snakemake_diamond")) - - class ShippedGraphsConform(unittest.TestCase): def test_every_shipped_graph(self): @@ -49,14 +31,6 @@ def test_every_shipped_graph(self): self.assertEqual(contract_violations(graph), []) -class ExtractedGraphsConform(unittest.TestCase): - - def test_every_extractor(self): - for name, graph in fixture_graphs(): - with self.subTest(extractor=name): - self.assertEqual(contract_violations(graph), []) - - class ContractCatchesBreakage(unittest.TestCase): def setUp(self): @@ -108,19 +82,14 @@ def test_engine_words_map_to_core_states(self): with self.subTest(word=word): self.assertEqual(task_status(word), status) - def test_every_extractor_emits_the_vocabulary(self): - for name, graph in fixture_graphs(): - with self.subTest(extractor=name): - for task in graph["tasks"].values(): - self.assertIn(task["status"], STATUSES) - class VersionMatchesPackaging(unittest.TestCase): - def test_init_and_pyproject_agree(self): + def test_installed_metadata_and_pyproject_agree(self): + # A checkout is installed editable; a stale install fails here, not in the field. text = (ROOT / "pyproject.toml").read_text() declared = re.search(r'^version = "([^"]+)"', text, re.M).group(1) - self.assertEqual(clew.__version__, declared) + self.assertEqual(version("clew-lineage"), declared) if __name__ == "__main__": diff --git a/tests/test_impact_guards.py b/tests/test_impact_guards.py index 8cf9402..5c737a3 100644 --- a/tests/test_impact_guards.py +++ b/tests/test_impact_guards.py @@ -14,7 +14,7 @@ import unittest from pathlib import Path -from clew.domains import nfcore +from clew.graph import results ROOT = Path(__file__).resolve().parent.parent @@ -61,7 +61,7 @@ class TestCleanedScratch(unittest.TestCase): def plan(self, tmp, *extra): graph, sheet = write_graph(tmp) out = run_impact("--graph", str(graph), "--samplesheet", str(sheet), - "--subject", "donor_001", "--json", "-", *extra).stdout + "--trigger", "patient:donor_001", "--json", "-", *extra).stdout return json.loads(out[out.index("{"):])["plan"][0] def test_gone_workdir_is_undetermined_until_results_are_checked(self): @@ -72,7 +72,7 @@ def test_gone_workdir_is_undetermined_until_results_are_checked(self): self.assertIsNone(item["action"]) self.assertIsNone(item["storage"]) self.assertIn("ALREADY_GONE", item["possible"]) - self.assertIn("published tree not checked", item["reason"]) + self.assertIn("published tree not checked", item["evidence"]) def test_gone_workdir_and_empty_results_is_already_gone(self): with tempfile.TemporaryDirectory() as tmp: @@ -92,7 +92,7 @@ def test_a_root_nothing_resolves_under_is_not_a_cleaned_run(self): results = Path(tmp, "results"); results.mkdir() graph, sheet = write_graph(tmp) result = run_impact("--graph", str(graph), "--samplesheet", str(sheet), - "--subject", "donor_001", "--json", "-", + "--trigger", "patient:donor_001", "--json", "-", "--work-root", str(work), "--results", str(results)) item = json.loads(result.stdout[result.stdout.index("{"):])["plan"][0] self.assertIsNone(item["action"]) @@ -117,35 +117,38 @@ def test_index_lists_directories_by_name(self): with tempfile.TemporaryDirectory() as tmp: Path(tmp, "a", "sample").mkdir(parents=True) Path(tmp, "a", "sample", "f.txt").write_text("abc") - index = nfcore.index_results(tmp) + index = results.index_results(tmp) self.assertEqual(index[("f.txt", 3)], ["a/sample/f.txt"]) - self.assertEqual(index[("sample", nfcore.DIRECTORY)], ["a/sample"]) + self.assertEqual(index[("sample", results.DIRECTORY)], ["a/sample"]) -class TestSubjectTrigger(unittest.TestCase): - """ - `--trigger subject:X` is the documented spelling of `--subject X`. It - used to bypass the domain adapter, so on an nf-core graph it found - nothing while the flag worked. - """ +class TestKindTrigger(unittest.TestCase): + """A kind the adapter declares resolves through the adapter, with its own flag.""" def test_with_a_samplesheet_it_is_the_withdrawal_path(self): with tempfile.TemporaryDirectory() as tmp: graph, sheet = write_graph(tmp) result = run_impact("--graph", str(graph), "--samplesheet", str(sheet), - "--trigger", "subject:donor_001", "--json", "-") + "--trigger", "patient:donor_001", "--json", "-") self.assertEqual(result.returncode, 0, result.stderr) payload = json.loads(result.stdout[result.stdout.index("{"):]) - self.assertEqual(payload["trigger"], "withdrawal of donor_001") + self.assertEqual(payload["trigger"], "removal of donor_001") self.assertEqual(payload["entry_tasks"], ["aa/000001"]) def test_without_a_samplesheet_the_message_points_at_one(self): with tempfile.TemporaryDirectory() as tmp: graph, _ = write_graph(tmp) - result = run_impact("--graph", str(graph), "--trigger", "subject:donor_001") + result = run_impact("--graph", str(graph), "--trigger", "patient:donor_001") + self.assertNotEqual(result.returncode, 0) + self.assertIn("--samplesheet is required", result.stderr) + + def test_a_kind_nobody_declares_is_refused_by_name(self): + with tempfile.TemporaryDirectory() as tmp: + graph, _ = write_graph(tmp) + result = run_impact("--graph", str(graph), "--trigger", "specimen:x") self.assertNotEqual(result.returncode, 0) - self.assertIn("carries a 'subject' label", result.stderr) - self.assertIn("--samplesheet", result.stderr) + self.assertIn("unknown trigger kind 'specimen'", result.stderr) + self.assertIn("patient", result.stderr) class TestGraphLimitsAreShown(unittest.TestCase): @@ -176,35 +179,35 @@ def test_subject_with_no_tagged_task_is_refused(self): sheet.write_text("patient,sample\ndonor_001,donor_001\n" "donor_999,donor_999\n") result = run_impact("--graph", str(graph), "--samplesheet", - str(sheet), "--subject", "donor_999") + str(sheet), "--trigger", "patient:donor_999") self.assertNotEqual(result.returncode, 0) self.assertIn("Not attributable", result.stderr) - self.assertNotIn("ids in the samplesheet", result.stderr) + self.assertNotIn("the ids and the run disagree", result.stderr) def test_total_mismatch_names_the_likely_cause(self): with tempfile.TemporaryDirectory() as tmp: graph, sheet = write_graph(tmp) sheet.write_text("patient,sample\nLIMS-0001,LIMS-0001\n") result = run_impact("--graph", str(graph), "--samplesheet", - str(sheet), "--subject", "LIMS-0001") + str(sheet), "--trigger", "patient:LIMS-0001") self.assertNotEqual(result.returncode, 0) - self.assertIn("ids in the samplesheet and the tags in the run disagree", - result.stderr) + self.assertIn("No task matched ANY patient", result.stderr) if __name__ == "__main__": unittest.main() -class TestSamplesheetAlone(unittest.TestCase): - """--samplesheet with no --subject prints the reach of every subject.""" +class TestBareKind(unittest.TestCase): + """--trigger kind with no value prints the reach of every value of that kind.""" - def test_the_table_lists_every_subject(self): + def test_the_table_lists_every_value(self): result = run_impact( "--graph", str(ROOT / "clew" / "data" / "graph5.json"), + "--trigger", "patient", "--samplesheet", str(ROOT / "clew" / "data" / "donors.csv")) self.assertEqual(result.returncode, 0, result.stderr) - self.assertIn("5 subjects", result.stdout) + self.assertIn("5 values", result.stdout) for subject in ("donor_001", "donor_003", "donor_005"): self.assertIn(subject, result.stdout) diff --git a/tests/test_impact_runs.py b/tests/test_impact_runs.py new file mode 100644 index 0000000..ea33f88 --- /dev/null +++ b/tests/test_impact_runs.py @@ -0,0 +1,55 @@ +"""impact --runs reads the engine's record and leaves the derived graph beside the plan.""" + +import io +import json +import shutil +import sys +import tempfile +import unittest +from contextlib import redirect_stdout, redirect_stderr +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from clew.questions import impact + +DATA = Path(__file__).resolve().parent.parent / "clew" / "data" + + +def run(*argv): + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + impact.main(list(argv)) + return out.getvalue() + + +class TestRuns(unittest.TestCase): + def setUp(self): + self.tmp = Path(tempfile.mkdtemp()) + shutil.copy(DATA / "graph5.json", self.tmp / "cohort.json") + + def tearDown(self): + shutil.rmtree(self.tmp) + + def test_a_directory_of_graphs_needs_no_extractor(self): + plan = self.tmp / "plan.json" + out = run("--runs", str(self.tmp), "--container", "gatk4", "--json", str(plan)) + self.assertIn("graph derived from", out) + derived = self.tmp / "plan.graph.json" + self.assertTrue(derived.exists()) + self.assertEqual(json.loads(derived.read_text())["run"]["name"], "cohort") + self.assertEqual(len(json.loads(plan.read_text())["plan"]), 68) + + def test_graph_and_runs_together_are_refused(self): + with self.assertRaises(SystemExit) as stop: + run("--graph", str(self.tmp / "cohort.json"), "--runs", str(self.tmp), "--container", "gatk4") + self.assertIn("not both", str(stop.exception)) + + def test_neither_is_refused(self): + with self.assertRaises(SystemExit) as stop: + run("--container", "gatk4") + self.assertIn("--graph or --runs", str(stop.exception)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_mcp_server.py b/tests/test_mcp_server.py index 78dd89b..9ba92c3 100644 --- a/tests/test_mcp_server.py +++ b/tests/test_mcp_server.py @@ -3,7 +3,7 @@ language model at a compliance record. Two things are tested here that are not really about MCP at all. That the -server is read-only — nothing it exposes can write a fact, and an auditor's +server is read-only, nothing it exposes can write a fact, and an auditor's chat session is the last place a new fact should be able to enter the record. And that bundles sealed from different logs are detected rather than silently interleaved into one plausible-looking history. @@ -41,16 +41,16 @@ def log_entry(seq, prev_hash, subject="s1", event_type="Withdrawn", def a_plan(): decision = policy_module.decide("REGENERABLE", storage="WRITABLE") return { - "clew_plan_version": 1, "trigger": "withdrawal of s1", + "clew_plan_version": 2, "trigger": "removal of s1", **policy_module.identify(policy_module.DEFAULT), "tasks_total": 10, "tasks_affected": 1, "entry_tasks": ["t0"], "actions": {decision["action"]: 1}, "plan": [{ "task": "t1", "process": "P", "name": "t1", "action": decision["action"], "rule": decision["rule"], - "because": decision["because"], "contribution": "REGENERABLE", - "storage": "WRITABLE", "exclusive": False, "terminal": False, - "reason": "test", "evidence_path": ["t0", "t1"], + "reason": decision["reason"], "contribution": "REGENERABLE", + "storage": "WRITABLE", "scope": "shared", "released": False, + "evidence": "test", "evidence_path": ["t0", "t1"], }], "caveats": [], } @@ -153,8 +153,8 @@ def converse_raw(self, lines, bundles): def test_malformed_input_does_not_kill_the_session(self): # Both shapes of bad line: unparseable, and valid JSON that is not an - # object. The second one is the dangerous one — it parses, and then - # has no .get() — and it must not end a session an auditor is in the + # object. The second one is the dangerous one, it parses, and then + # has no .get(), and it must not end a session an auditor is in the # middle of. replies = self.converse_raw( ["{ not json at all\n", diff --git a/tests/test_mode.py b/tests/test_mode.py index 6985038..3decd6c 100644 --- a/tests/test_mode.py +++ b/tests/test_mode.py @@ -1,6 +1,6 @@ """ Selector × mode: the same subject trigger must produce different verdicts -under remove (withdrawal) and distrust (contamination), and remove must be +under remove (withdrawal) and trace (contamination), and remove must be refused for non-subject selectors. """ @@ -29,7 +29,7 @@ def plan(self, *extra): return json.loads(out[out.index("{"):])["plan"] # NOTE these assert the exclusive/terminal FACTS, not the final actions. - # Actions also depend on storage — live disk state — so pinning them + # Actions also depend on storage, live disk state, so pinning them # made the suite fail the day work/ was (correctly) cleaned up. The # facts→action table itself is pinned hermetically in test_policy. # @@ -43,13 +43,13 @@ def outcomes(self, item): return [item["action"]] if item["action"] else list(item["possible"]) def test_withdrawal_marks_exclusive(self): - plan = self.plan("--donor", "ERR10000000") - exclusive = [i for i in plan if i["exclusive"]] + plan = self.plan("--trigger", "sample:ERR10000000") + exclusive = [i for i in plan if i["scope"] == "exclusive"] self.assertEqual(len(exclusive), 41) self.assertEqual(len(plan), 46) # Exclusive artifacts must never resolve to a rebuild: they either # get destroyed, are already gone, or are quarantined unwritable. - # True whether or not storage has been checked — an unverified item + # True whether or not storage has been checked, an unverified item # must not have a rebuild among its possibilities either. for i in exclusive: for outcome in self.outcomes(i): @@ -57,18 +57,18 @@ def test_withdrawal_marks_exclusive(self): ("DESTROY", "ALREADY_GONE", "QUARANTINE"), i["task"]) def test_contamination_marks_nothing_exclusive(self): - # Same specimen, same radius — but the data is wrong, not withdrawn, + # Same specimen, same radius, but the data is wrong, not withdrawn, # so nothing is owned-and-destroyable. - plan = self.plan("--donor", "ERR10000000", "--mode", "distrust") + plan = self.plan("--trigger", "sample:ERR10000000", "--mode", "trace") self.assertEqual(len(plan), 46) - self.assertEqual([i for i in plan if i["exclusive"]], []) + self.assertEqual([i for i in plan if i["scope"] == "exclusive"], []) for i in plan: self.assertNotIn("DESTROY", self.outcomes(i), i["task"]) def test_remove_refused_for_input_selector(self): result = run_blast("--input", "nCoV-2019.primer.bed", "--mode", "remove") self.assertNotEqual(result.returncode, 0) - self.assertIn("requires a subject trigger", result.stderr) + self.assertIn("needs a kind that owns something", result.stderr) if __name__ == "__main__": diff --git a/tests/test_pending.py b/tests/test_pending.py new file mode 100644 index 0000000..a17c001 --- /dev/null +++ b/tests/test_pending.py @@ -0,0 +1,97 @@ +"""Unattended runs: the adapter says what is pending, impact answers each.""" + +import io +import sys +import tempfile +import unittest +from contextlib import redirect_stdout, redirect_stderr +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from clew.contracts import Adapter +from clew.contracts.adapter import check +from clew.provider.nextflow import adapter_sarek as sarek +from clew.questions import impact + +DATA = Path(__file__).resolve().parent.parent / "clew" / "data" +GRAPH = str(DATA / "graph5.json") +SHEET = str(DATA / "donors.csv") + + +class Site(sarek.Sarek): + """Sarek plus what a site adds: where the sheet is, what is pending.""" + name = "sarek-test-site" + queue = [] + triggers = {"patient": sarek.base.SheetKind("patient", members="sample", + locate=lambda graph: SHEET)} + + def pending(self): + return list(self.queue) + + +def run(*argv): + out, err = io.StringIO(), io.StringIO() + with redirect_stdout(out), redirect_stderr(err): + code = impact.main(["--pipeline", "sarek-test-site", *argv]) + return code, out.getvalue(), err.getvalue() + + +class TestPending(unittest.TestCase): + @classmethod + def tearDownClass(cls): + Adapter.registered.pop("sarek-test-site", None) + + def test_each_pending_trigger_is_answered(self): + Site.queue = [ + {"kind": "patient", "value": "donor_003", "asserted_by": "qa", "date": "2026-09-10"}, + {"kind": "container", "value": "gatk4"}, + ] + code, out, _ = run("--graph", GRAPH) + self.assertEqual(code, 0) + self.assertIn("TRIGGER patient:donor_003 asserted by qa on 2026-09-10", out) + self.assertIn("TRIGGER container:gatk4", out) + self.assertIn("2 triggers, 0 failed", out) + + def test_a_bad_trigger_fails_its_answer_not_the_batch(self): + Site.queue = [{"kind": "container", "value": "no-such-image"}, + {"kind": "container", "value": "gatk4"}] + code, out, err = run("--graph", GRAPH) + self.assertEqual(code, 1) + self.assertIn("FAILED container:no-such-image", err) + self.assertIn("2 triggers, 1 failed", out) + + def test_nothing_pending_and_nothing_asked_is_refused(self): + Site.queue = [] + with self.assertRaises(SystemExit) as stop: + run("--graph", GRAPH) + self.assertIn("nothing to ask", str(stop.exception)) + + def test_an_explicit_question_wins_over_pending(self): + Site.queue = [{"kind": "container", "value": "gatk4"}] + code, out, _ = run("--graph", GRAPH, "--container", "bcftools") + self.assertIsNone(code) + self.assertNotIn("triggers, ", out) + + def test_each_answer_gets_its_own_output_file(self): + Site.queue = [{"kind": "container", "value": "gatk4"}] + with tempfile.TemporaryDirectory() as tmp: + plan = Path(tmp) / "plan.json" + code, _, _ = run("--graph", GRAPH, "--json", str(plan)) + self.assertEqual(code, 0) + self.assertTrue((Path(tmp) / "plan.container-gatk4.json").exists()) + + def test_malformed_trigger_is_refused_before_anything_runs(self): + self.assertEqual(check({"kind": "subject", "value": "x"}), []) + self.assertTrue(check({"kind": "subject"})) + self.assertTrue(check("subject:x")) + + def test_adapter_can_find_the_sheet_itself(self): + Site.queue = [] + code, out, _ = run("--graph", GRAPH, "--trigger", "patient:donor_003") + self.assertIsNone(code) + self.assertIn("donor_003", out) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_plan_cost.py b/tests/test_plan_cost.py new file mode 100644 index 0000000..c2ee18e --- /dev/null +++ b/tests/test_plan_cost.py @@ -0,0 +1,62 @@ +"""A plan sums whatever metrics the provider recorded, per verdict, and counts what is missing.""" + +import sys +import unittest +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from clew.graph.graph import contract_violations +from clew.questions.impact import plan_cost + + +def graph(): + def task(h, **extra): + return {"hash": h, "name": h, "process": h, "container": "img", "status": "COMPLETED", + "script": "run", "workdir": f"/w/{h}", **extra} + return { + "tasks": { + "a/1": task("a/1", metrics={"wall_s": 10.0, "kwh": 0.5}), + "b/2": task("b/2", metrics={"wall_s": 2.5}), + "c/3": task("c/3"), + }, + "edges": [], "outputs": {}, + } + + +def decided(*pairs): + return [(h, {}, {"action": a, "rule": "R", "reason": ""}) for h, a in pairs] + + +class TestPlanCost(unittest.TestCase): + def test_sums_each_metric_per_verdict_and_counts_the_gaps(self): + cost = plan_cost(graph(), decided(("a/1", "REGENERATE"), ("b/2", "REGENERATE"), + ("c/3", "REGENERATE"))) + bucket = cost["by_action"]["REGENERATE"] + self.assertEqual(bucket["tasks"], 3) + self.assertEqual(bucket["metrics"], {"wall_s": 12.5, "kwh": 0.5}) + self.assertEqual(bucket["missing"], {"wall_s": 1, "kwh": 2}) + + def test_verdicts_are_kept_apart(self): + cost = plan_cost(graph(), decided(("a/1", "REGENERATE"), ("b/2", "QUARANTINE"))) + self.assertEqual(cost["by_action"]["REGENERATE"]["metrics"], {"wall_s": 10.0, "kwh": 0.5}) + self.assertEqual(cost["by_action"]["QUARANTINE"]["metrics"], {"wall_s": 2.5}) + + def test_the_engine_names_no_metric(self): + import clew.questions.impact as module + import clew.graph.graph as graph_module + for word in ("duration_s", "cpus", "price", "memory_gb"): + self.assertNotIn(word, Path(module.__file__).read_text()) + self.assertNotIn(word, Path(graph_module.__file__).read_text()) + + def test_the_contract_accepts_numbers_and_refuses_the_rest(self): + g = graph() + self.assertEqual(contract_violations(g), []) + g["tasks"]["a/1"]["metrics"] = {"wall_s": "10"} + self.assertTrue(any("metrics" in p for p in contract_violations(g))) + g["tasks"]["a/1"]["metrics"] = {"wall_s": -1} + self.assertTrue(any("metrics" in p for p in contract_violations(g))) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_plan_json.py b/tests/test_plan_json.py index 8f6738d..7c5b61b 100644 --- a/tests/test_plan_json.py +++ b/tests/test_plan_json.py @@ -43,10 +43,10 @@ def test_the_plan_names_the_policy_it_was_computed_under(self): def test_every_item_states_its_basis(self): # A decided item cites the rule that decided it. An undecided one has - # no rule to cite, and must instead carry the candidates — so that a + # no rule to cite, and must instead carry the candidates, so that a # missing action is never mistakable for "nothing to do". for item in self.payload["plan"]: - self.assertTrue(item["because"], item["task"]) + self.assertTrue(item["reason"], item["task"]) if item["action"]: self.assertTrue(item["rule"], item["task"]) self.assertNotIn("possible", item) @@ -55,7 +55,7 @@ def test_every_item_states_its_basis(self): self.assertGreater(len(item["possible"]), 1, item["task"]) def test_the_cited_rule_actually_yields_the_stated_action(self): - # Guards against the citation drifting from the verdict — a plan whose + # Guards against the citation drifting from the verdict, a plan whose # rule ids are decorative would be worse than one with none. from clew.ledger import policy by_id = {r["id"]: r for r in policy.DEFAULT["rules"]} @@ -77,7 +77,7 @@ def test_this_run_is_undetermined_because_no_work_root_was_given(self): def test_shape_and_counts(self): p = self.payload - self.assertEqual(p["clew_plan_version"], 1) + self.assertEqual(p["clew_plan_version"], 2) self.assertEqual(p["trigger"], "container:ivar") self.assertEqual(p["tasks_total"], 219) self.assertEqual(p["tasks_affected"], 160) @@ -87,7 +87,7 @@ def test_shape_and_counts(self): def test_regenerable_items_carry_recorded_evidence(self): # Whether a task's action is REGENERATE or ALREADY_GONE depends on # live disk state; what must ALWAYS hold is that REGENERABLE tasks - # carry the recorded script and container — that recording is what + # carry the recorded script and container, that recording is what # the classification was based on. regen = [i for i in self.payload["plan"] if i["contribution"] == "REGENERABLE"] diff --git a/tests/test_policy.py b/tests/test_policy.py index d7eb379..e9c1fea 100644 --- a/tests/test_policy.py +++ b/tests/test_policy.py @@ -1,8 +1,7 @@ """ The versioned remediation table. -Two jobs here. The first is the decision table itself, tested exhaustively — -these assertions moved over from test_contribution.py unchanged when the +Two jobs here. The first is the decision table itself, tested exhaustively, these assertions moved over from test_contribution.py unchanged when the table became data, which is the point: turning an if-ladder into rules had to change nothing about what Clew decides. @@ -24,25 +23,24 @@ from clew.ledger import policy as p # Every combination the engine can present, including classes it should never -# see. 4 x 3 x 2 x 2. +# see. 4 x 3 x 2 x 2 x 2. SPACE = list(product(list(c.CLASSES) + ["GARBAGE"], c.STORAGE, - (True, False), (True, False))) + (True, False), (True, False), p.MODES)) class TestDecisionTable(unittest.TestCase): """Unchanged semantics. Same assertions as before the table became data.""" def test_destroyed_storage_wins_over_everything_except_publication(self): - # The v1 -> v2 change. Deleting your copy of a published artifact does - # not un-publish it, so the disclosure obligation outlives the bytes. + # Deleting your copy of a released artifact does not un-release it. for klass, exclusive in product(c.CLASSES, (True, False)): self.assertEqual( - p.remediate(klass, storage=c.DESTROYED, exclusive=exclusive, - terminal=False), + p.remediate(klass, storage=c.DESTROYED, scope=(p.EXCLUSIVE if exclusive else p.SHARED), + released=False), c.ALREADY_GONE) self.assertEqual( - p.remediate(klass, storage=c.DESTROYED, exclusive=exclusive, - terminal=True), + p.remediate(klass, storage=c.DESTROYED, scope=(p.EXCLUSIVE if exclusive else p.SHARED), + released=True), c.NOTIFY_ONLY) def test_terminal_wins_over_class_and_exclusivity(self): @@ -50,22 +48,23 @@ def test_terminal_wins_over_class_and_exclusivity(self): # contribution is. Remediation stops; notification does not. for klass, exclusive in product(c.CLASSES, (True, False)): self.assertEqual( - p.remediate(klass, exclusive=exclusive, terminal=True), + p.remediate(klass, scope=(p.EXCLUSIVE if exclusive else p.SHARED), released=True), c.NOTIFY_ONLY) def test_exclusive_writable_is_destroyed(self): for klass in c.CLASSES: - self.assertEqual(p.remediate(klass, exclusive=True), c.DESTROY) + self.assertEqual(p.remediate(klass, scope=p.EXCLUSIVE), c.DESTROY) def test_exclusive_worm_is_quarantined(self): for klass in c.CLASSES: self.assertEqual( - p.remediate(klass, storage=c.WORM, exclusive=True), + p.remediate(klass, storage=c.WORM, scope=p.EXCLUSIVE), c.QUARANTINE) def test_separable_shared(self): - self.assertEqual(p.remediate(c.SEPARABLE), c.PURGE) - self.assertEqual(p.remediate(c.SEPARABLE, storage=c.WORM), c.REGENERATE) + self.assertEqual(p.remediate(c.SEPARABLE, mode=p.REMOVE), c.PURGE) + self.assertEqual(p.remediate(c.SEPARABLE, storage=c.WORM, mode=p.REMOVE), + c.REGENERATE) def test_regenerable_shared(self): self.assertEqual(p.remediate(c.REGENERABLE), c.REGENERATE) @@ -81,9 +80,9 @@ def test_unknown_class_is_quarantined_not_purged(self): self.assertEqual(p.remediate("MYSTERY"), c.QUARANTINE) def test_every_combination_yields_one_known_action_and_a_rule(self): - for klass, storage, exclusive, terminal in SPACE: - decision = p.decide(klass, storage=storage, exclusive=exclusive, - terminal=terminal) + for klass, storage, exclusive, terminal, mode in SPACE: + decision = p.decide(klass, storage=storage, scope=(p.EXCLUSIVE if exclusive else p.SHARED), + released=terminal, mode=mode) self.assertIn(decision["action"], p.ACTIONS) self.assertTrue(decision["rule"]) self.assertNotEqual(c.explain(decision["action"]), "unknown action") @@ -93,16 +92,16 @@ class TestRulesAreReachable(unittest.TestCase): def test_no_rule_is_dead(self): # A rule that can never match is indistinguishable from a deleted one, # except that the file still shows it and everyone believes it applies. - reached = {p.decide(k, storage=s, exclusive=e, terminal=t)["rule"] - for k, s, e, t in SPACE} - declared = {rule["id"] for rule in p.V1["rules"]} + reached = {p.decide(k, storage=s, scope=(p.EXCLUSIVE if e else p.SHARED), released=t, mode=m)["rule"] + for k, s, e, t, m in SPACE} + declared = {rule["id"] for rule in p.DEFAULT["rules"]} self.assertEqual(declared - reached, set()) def test_the_builtin_table_never_falls_through(self): # The fallthrough guard is for policies that are wrong. Ours must not # be relying on it. - reached = {p.decide(k, storage=s, exclusive=e, terminal=t)["rule"] - for k, s, e, t in SPACE} + reached = {p.decide(k, storage=s, scope=(p.EXCLUSIVE if e else p.SHARED), released=t, mode=m)["rule"] + for k, s, e, t, m in SPACE} self.assertNotIn(p.FALLTHROUGH_RULE, reached) @@ -122,11 +121,11 @@ def test_changing_only_a_rationale_changes_the_hash(self): # an assessor reads, and editing it changes what the organisation is # on record as having meant. altered = json.loads(json.dumps(p.V1)) - altered["rules"][0]["because"] = "because I said so" + altered["rules"][0]["reason"] = "because I said so" self.assertNotEqual(p.fingerprint(p.V1), p.fingerprint(altered)) def test_reordering_rules_changes_the_hash(self): - # Order is semantics here — first match wins. + # Order is semantics here, first match wins. altered = json.loads(json.dumps(p.V1)) altered["rules"][0], altered["rules"][1] = (altered["rules"][1], altered["rules"][0]) @@ -182,7 +181,7 @@ def test_duplicate_rule_ids_are_rejected(self): def test_a_rule_without_a_rationale_is_rejected(self): bad = self.valid() - bad["rules"][0]["because"] = " " + bad["rules"][0]["reason"] = " " self.assertRejected(bad, "rationale") def test_the_fallthrough_name_is_reserved(self): @@ -206,15 +205,14 @@ class TestVersionsAreImmutable(unittest.TestCase): A shipped version is a historical record, not a place to fix things. Plans cite a version and a hash. If a version can be edited in place, the - label is a lie and the hash proves nothing — a January plan would replay + label is a lie and the hash proves nothing, a January plan would replay under a table nobody had in January. So the hashes are frozen here as literals: editing a shipped policy fails this test, and the fix is to add a version, never to change one. """ FROZEN = { - "v1": "dbb59de6d85fc0f87f4bc7d490b6ce34f8bf9222875386a721ec4e18f3ee0461", - "v2": "e6ba60ffe6763949106eca86f7888c3cc3e28c920fd999ce275d708153152642", + "v1": "f1f49f91c8a49f7e80eee15961796fb777a0e6f94783b19c8d79d54616275100", } def test_shipped_hashes_have_not_moved(self): @@ -226,35 +224,35 @@ def test_shipped_hashes_have_not_moved(self): f"plans citing it stay replayable.") def test_every_registered_version_is_frozen_here(self): - # Adding a version without freezing its hash would leave it editable. self.assertEqual(set(p.REGISTRY), set(self.FROZEN)) - def test_v1_and_v2_are_genuinely_different_tables(self): - self.assertNotEqual(p.fingerprint(p.V1), p.fingerprint(p.V2)) + def test_rule_ids_follow_table_order(self): + self.assertEqual([r["id"] for r in p.V1["rules"]], + [f"R{i}" for i in range(1, 10)]) - def test_rule_ids_are_stable_across_versions(self): - # Ids identify rules, not positions, so two plans on different - # versions stay comparable line by line. - self.assertEqual({r["id"] for r in p.V1["rules"]}, - {r["id"] for r in p.V2["rules"]}) - def test_only_the_order_of_r1_and_r2_changed(self): - self.assertEqual([r["id"] for r in p.V1["rules"]], - ["R1", "R2", "R3", "R4", "R5", "R6", "R7", "R8"]) - self.assertEqual([r["id"] for r in p.V2["rules"]], - ["R2", "R1", "R3", "R4", "R5", "R6", "R7", "R8"]) - - def test_v1_still_decides_the_way_it_always_did(self): - # The point of keeping it: a plan from before the change replays - # under the table that produced it, not under today's. - self.assertEqual( - p.remediate(c.REGENERABLE, storage=c.DESTROYED, terminal=True, - policy=p.resolve("v1")), - c.ALREADY_GONE) - self.assertEqual( - p.remediate(c.REGENERABLE, storage=c.DESTROYED, terminal=True, - policy=p.resolve("v2")), - c.NOTIFY_ONLY) +class TestMode(unittest.TestCase): + """A removed subject's separable part is purged; a corrected one is recomputed.""" + + def test_removed_is_purged_and_corrected_is_regenerated(self): + gone = p.decide(c.SEPARABLE, mode=p.REMOVE) + changed = p.decide(c.SEPARABLE, mode=p.TRACE) + self.assertEqual((gone["action"], gone["rule"]), (c.PURGE, "R6")) + self.assertEqual((changed["action"], changed["rule"]), (c.REGENERATE, "R5")) + + def test_mode_matters_only_for_a_separable_writable_shared_artifact(self): + for klass, storage, exclusive, terminal, _ in SPACE: + scope = p.EXCLUSIVE if exclusive else p.SHARED + a = p.remediate(klass, storage=storage, scope=scope, released=terminal, mode=p.REMOVE) + b = p.remediate(klass, storage=storage, scope=scope, released=terminal, mode=p.TRACE) + differs = (klass == c.SEPARABLE and storage == c.WRITABLE + and not exclusive and not terminal) + self.assertEqual(a != b, differs, (klass, storage, exclusive, terminal)) + + def test_an_unknown_mode_is_withheld_not_guessed(self): + decision = p.decide(c.SEPARABLE, mode=None) + self.assertIsNone(decision["action"]) + self.assertEqual(decision["possible"], {c.PURGE: "R6", c.REGENERATE: "R5"}) class TestLoadAndResolve(unittest.TestCase): @@ -269,13 +267,12 @@ def test_a_bad_file_is_refused_at_load(self): with tempfile.TemporaryDirectory() as tmp: path = Path(tmp) / "policy.json" path.write_text(json.dumps({"version": "x", "rules": [ - {"id": "R1", "when": {}, "action": "SHRED", "because": "no"}]})) + {"id": "R1", "when": {}, "action": "SHRED", "reason": "no"}]})) with self.assertRaises(p.InvalidPolicy): p.load(path) def test_a_shipped_version_resolves(self): self.assertIs(p.resolve("v1"), p.V1) - self.assertIs(p.resolve("v2"), p.V2) def test_an_unknown_version_raises_rather_than_substituting(self): # A plan citing a version this build does not have cannot be replayed @@ -291,7 +288,7 @@ def test_no_match_falls_through_to_quarantine(self): "version": "narrow", "rules": [p.rule("ONLY", c.PURGE, "matches almost nothing", contribution=c.SEPARABLE, storage=c.WORM, - exclusive=True, terminal=True)], + scope=p.EXCLUSIVE, released=True)], }) decision = p.decide(c.REGENERABLE, policy=narrow) self.assertEqual(decision["action"], c.QUARANTINE) @@ -342,9 +339,8 @@ class TestUnverifiedStorage(unittest.TestCase): """ storage=None means NOT CHECKED, which is not one of the three values. - The old behaviour returned DESTROYED whenever a workdir did not resolve — - on another host, with an unmounted volume, from an extractor that records - no workdir at all — and DESTROYED becomes ALREADY_GONE, "nothing to do". + The old behaviour returned DESTROYED whenever a workdir did not resolve, on another host, with an unmounted volume, from an extractor that records + no workdir at all, and DESTROYED becomes ALREADY_GONE, "nothing to do". That is the one error direction this project exists not to make, so an unchecked dimension now yields no verdict rather than a convenient one. """ @@ -358,7 +354,7 @@ def test_disagreement_yields_no_action(self): def test_the_possible_map_names_the_rule_for_each_candidate(self): possible = p.decide(c.REGENERABLE, storage=None, - exclusive=True)["possible"] + scope=p.EXCLUSIVE)["possible"] by_id = {r["id"]: r for r in p.V1["rules"]} for action, rule_id in possible.items(): self.assertEqual(by_id[rule_id]["action"], action) @@ -375,36 +371,22 @@ def test_agreement_across_all_storage_states_is_a_real_answer(self): self.assertEqual(decision["action"], c.QUARANTINE) self.assertEqual(decision["rule"], "Q") self.assertNotIn("possible", decision) - self.assertIn("every possible state", decision["because"]) - - def test_under_v1_nothing_is_decidable_without_checking_storage(self): - # A consequence of R1 being first: DESTROYED always yields - # ALREADY_GONE, and no other rule can, so every unchecked item - # disagrees with itself. Under v1, storage MUST be verified — which - # is half of why v2 exists. - for klass, exclusive, terminal in product(c.CLASSES, (True, False), - (True, False)): - decision = p.decide(klass, storage=None, exclusive=exclusive, - terminal=terminal, policy=p.V1) - self.assertIsNone(decision["action"], - f"{klass} {exclusive} {terminal}") - - def test_under_v2_publication_needs_no_disk_check(self): - # The verdict genuinely does not depend on the storage state, so it - # is returned rather than withheld. Not a guess: all three possible - # states give the same answer. + self.assertIn("every possible state", decision["reason"]) + + def test_a_released_artifact_needs_no_disk_check(self): + # Release is asked before existence, so the verdict does not depend + # on storage and is returned rather than withheld. for klass, exclusive in product(c.CLASSES, (True, False)): - decision = p.decide(klass, storage=None, exclusive=exclusive, - terminal=True, policy=p.V2) + decision = p.decide(klass, storage=None, + scope=(p.EXCLUSIVE if exclusive else p.SHARED), released=True) self.assertEqual(decision["action"], c.NOTIFY_ONLY) - self.assertEqual(decision["rule"], "R2") + self.assertEqual(decision["rule"], "R1") - def test_under_v2_unpublished_items_still_need_a_disk_check(self): - # v2 narrows what must be verified; it does not remove the need. + def test_an_unreleased_artifact_still_needs_a_disk_check(self): for klass, exclusive in product(c.CLASSES, (True, False)): self.assertIsNone( - p.decide(klass, storage=None, exclusive=exclusive, - terminal=False, policy=p.V2)["action"]) + p.decide(klass, storage=None, + scope=(p.EXCLUSIVE if exclusive else p.SHARED), released=False)["action"]) def test_undetermined_is_not_an_action(self): # Code iterating the action set must not find it there and start @@ -425,43 +407,42 @@ class TestEveryDimensionIsChecked(unittest.TestCase): """ def test_an_unverified_terminal_is_evaluated_like_storage(self): - decision = p.decide(c.REGENERABLE, storage=c.WRITABLE, terminal=None) + decision = p.decide(c.REGENERABLE, storage=c.WRITABLE, released=None) self.assertIsNone(decision["action"]) self.assertEqual(sorted(decision["possible"]), [c.NOTIFY_ONLY, c.REGENERATE]) - self.assertIn("terminal", decision["because"]) + self.assertIn("released", decision["reason"]) def test_an_unverified_exclusive_is_evaluated_too(self): - decision = p.decide(c.REGENERABLE, storage=c.WRITABLE, exclusive=None) + decision = p.decide(c.REGENERABLE, storage=c.WRITABLE, scope=None) self.assertEqual(sorted(decision["possible"]), [c.DESTROY, c.REGENERATE]) def test_several_unverified_dimensions_combine(self): - decision = p.decide(c.REGENERABLE, storage=None, exclusive=None, - terminal=None) + decision = p.decide(c.REGENERABLE, storage=None, scope=None, + released=None) self.assertIsNone(decision["action"]) self.assertIn(c.NOTIFY_ONLY, decision["possible"]) self.assertIn(c.ALREADY_GONE, decision["possible"]) self.assertIn(c.DESTROY, decision["possible"]) - def test_agreement_over_an_unverified_terminal_is_a_real_answer(self): - # Published or not, a DESTROYED artifact under v1 is ALREADY_GONE. - decision = p.decide(c.REGENERABLE, storage=c.DESTROYED, terminal=None, - policy=p.V1) - self.assertEqual(decision["action"], c.ALREADY_GONE) - self.assertIn("terminal unverified", decision["because"]) + def test_agreement_over_an_unverified_mode_is_a_real_answer(self): + # Removed or corrected, a regenerable artifact is regenerated. + decision = p.decide(c.REGENERABLE, mode=None) + self.assertEqual(decision["action"], c.REGENERATE) + self.assertIn("mode unverified", decision["reason"]) def test_a_value_outside_the_dimension_is_an_error(self): with self.assertRaises(ValueError): p.decide(c.REGENERABLE, storage="writable") with self.assertRaises(ValueError): - p.decide(c.REGENERABLE, terminal="yes") + p.decide(c.REGENERABLE, released="yes") with self.assertRaises(ValueError): - p.decide(c.REGENERABLE, exclusive="no") + p.decide(c.REGENERABLE, scope="no") def test_an_int_does_not_pass_as_a_bool(self): with self.assertRaises(ValueError): - p.decide(c.REGENERABLE, terminal=1) + p.decide(c.REGENERABLE, released=1) def test_an_unrecognised_class_still_fails_closed_rather_than_erroring(self): # The one dimension with a defined fallback keeps it. @@ -478,14 +459,15 @@ def test_the_same_inputs_under_two_policies_differ_visibly(self): "this organisation does not purge in place", contribution=c.SEPARABLE)], }) - self.assertEqual(p.remediate(c.SEPARABLE), c.PURGE) - self.assertEqual(p.remediate(c.SEPARABLE, policy=strict), c.QUARANTINE) + self.assertEqual(p.remediate(c.SEPARABLE, mode=p.REMOVE), c.PURGE) + self.assertEqual(p.remediate(c.SEPARABLE, mode=p.REMOVE, policy=strict), + c.QUARANTINE) def test_decisions_are_stable_across_repeated_calls(self): - first = [p.decide(k, storage=s, exclusive=e, terminal=t) - for k, s, e, t in SPACE] - second = [p.decide(k, storage=s, exclusive=e, terminal=t) - for k, s, e, t in SPACE] + first = [p.decide(k, storage=s, scope=(p.EXCLUSIVE if e else p.SHARED), released=t, mode=m) + for k, s, e, t, m in SPACE] + second = [p.decide(k, storage=s, scope=(p.EXCLUSIVE if e else p.SHARED), released=t, mode=m) + for k, s, e, t, m in SPACE] self.assertEqual(first, second) diff --git a/tests/test_policy_providers.py b/tests/test_policy_providers.py new file mode 100644 index 0000000..6f641f7 --- /dev/null +++ b/tests/test_policy_providers.py @@ -0,0 +1,73 @@ +"""A provider may register its own policy table by name, and override a shipped one.""" + +import copy +import json +import sys +import tempfile +import unittest +from pathlib import Path +from unittest import mock + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from clew.ledger import policy + + +class FakeEntry: + def __init__(self, value): + self.value = value + + def load(self): + return self.value + + +def table(version, **changes): + t = copy.deepcopy(policy.V1) + t["version"] = version + t.update(changes) + return t + + +def registered(*entries): + """Patch the entry-point reader with (name, dist, entry) rows.""" + return mock.patch("clew.contracts.registry.entry_points", return_value=list(entries)) + + +class TestPolicyProviders(unittest.TestCase): + def test_a_registered_table_resolves_by_name(self): + with registered(("qbc-v1", "clew-qbc", FakeEntry(table("qbc-v1")))): + found = policy.resolve_or_load("qbc-v1") + self.assertEqual(found["version"], "qbc-v1") + self.assertEqual(policy.identify(found)["policy_version"], "qbc-v1") + + def test_a_table_may_be_a_json_file(self): + with tempfile.TemporaryDirectory() as tmp: + path = Path(tmp) / "site.json" + path.write_text(json.dumps(table("site-v3"))) + with registered(("site-v3", "clew-site", FakeEntry(str(path)))): + self.assertEqual(policy.resolve("site-v3")["version"], "site-v3") + + def test_a_registered_name_overrides_a_shipped_version(self): + mine = table("v1", description="the site's own v1") + with registered(("v1", "clew-site", FakeEntry(mine))): + self.assertEqual(policy.resolve("v1")["description"], "the site's own v1") + self.assertNotEqual(policy.resolve("v1")["description"], "the site's own v1") + + def test_an_invalid_table_is_refused_naming_the_provider(self): + broken = table("bad-v1") + broken["rules"][0]["action"] = "EXPLODE" + with registered(("bad-v1", "clew-site", FakeEntry(broken))): + with self.assertRaises(policy.InvalidPolicy) as stop: + policy.available() + self.assertIn("clew-site", str(stop.exception)) + self.assertIn("EXPLODE", str(stop.exception)) + + def test_name_and_version_must_agree(self): + with registered(("qbc-v1", "clew-qbc", FakeEntry(table("qbc-v9")))): + with self.assertRaises(policy.InvalidPolicy) as stop: + policy.available() + self.assertIn("must agree", str(stop.exception)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_providers.py b/tests/test_providers.py new file mode 100644 index 0000000..13e25be --- /dev/null +++ b/tests/test_providers.py @@ -0,0 +1,28 @@ +"""clew providers: every domain and extractor installed, and where each came from.""" + +import io +import sys +import unittest +from contextlib import redirect_stdout +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent.parent)) + +from clew import providers + + +class TestProviders(unittest.TestCase): + def test_builtins_are_listed_with_their_package(self): + with redirect_stdout(io.StringIO()) as out: + code = providers.main([]) + text = out.getvalue() + self.assertEqual(code, 0) + self.assertIn("clew.adapters", text) + self.assertIn("clew.extractors", text) + for name, package in (("sarek", "clew-nextflow"), ("snakemake", "clew-snakemake"), + ("nextflow", "clew-nextflow"), ("cromwell", "clew-cromwell")): + self.assertRegex(text, rf"\n {name}\s+{package}") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_query.py b/tests/test_query.py index 8c43cdb..ad6c8fe 100644 --- a/tests/test_query.py +++ b/tests/test_query.py @@ -2,7 +2,7 @@ The query surface: what an auditor's question resolves to. One rule dominates these tests. An answer that asserts something must carry -citations, and there must be no way to get one that does not — because the +citations, and there must be no way to get one that does not, because the consumer on the other side is a language model whose fluent, confident prose an auditor cannot distinguish from an accurate one by reading it. Citations are what make a bad paraphrase checkable instead of persuasive. @@ -37,7 +37,7 @@ def entry(seq, subject, event_type, effective_from, body=None, def plan_with(items, trigger="test:trigger"): return { - "clew_plan_version": 1, "trigger": trigger, + "clew_plan_version": 2, "trigger": trigger, **policy_module.identify(policy_module.DEFAULT), "tasks_total": 10, "tasks_affected": len(items), "entry_tasks": ["t0"], "plan": items, "caveats": ["a stated limit"], @@ -47,9 +47,9 @@ def plan_with(items, trigger="test:trigger"): def item(task, action="REGENERATE", rule="R7", **overrides): base = { "task": task, "process": "P", "name": task, "action": action, - "rule": rule, "because": "because", "contribution": "REGENERABLE", - "storage": "WRITABLE", "exclusive": False, "terminal": False, - "reason": "test", "evidence_path": ["t0", task], + "rule": rule, "reason": "because", "contribution": "REGENERABLE", + "storage": "WRITABLE", "scope": "shared", "released": False, + "evidence": "test", "evidence_path": ["t0", task], } base.update(overrides) return base @@ -86,7 +86,7 @@ def test_a_verdict_cites_the_rule_with_its_own_rationale(self): if c["kind"] == "policy_rule") self.assertEqual(rule["rule"], "R7") self.assertEqual(rule["policy_version"], policy_module.DEFAULT["version"]) - self.assertTrue(rule["because"]) + self.assertTrue(rule["reason"]) class TestEmptyIsNotClean(unittest.TestCase): diff --git a/tests/test_report.py b/tests/test_report.py index 90c02b5..10f84dc 100644 --- a/tests/test_report.py +++ b/tests/test_report.py @@ -16,28 +16,28 @@ from clew.views import report PLAN = { - "clew_plan_version": 1, + "clew_plan_version": 2, "trigger": "input:reference.dat", "tasks_affected": 3, "tasks_total": 8, "entry_tasks": ["run/a"], - "policy_version": "v2", + "policy_version": "v1", "policy_hash": "e6ba60ffe6763949106eca86f7888c3cc", "actions": {"REGENERATE": 2}, "caveats": ["uninstrumented systems are unknown, never clean"], "plan": [ {"task": "run/a", "process": "PREP", "target": "ssh://gpu-box", "contribution": "REGENERABLE", "storage": None, "action": None, - "possible": "REGENERATE", "reason": "storage not checked", - "exclusive": False, "terminal": False, "rule": "r1"}, + "possible": "REGENERATE", "evidence": "storage not checked", + "scope": "shared", "released": False, "rule": "r1"}, {"task": "run/b", "process": "RUN", "target": "ssh://gpu-box", "contribution": "REGENERABLE", "storage": "WRITABLE", - "action": "REGENERATE", "reason": "can be re-executed", - "exclusive": False, "terminal": False, "rule": "r1"}, + "action": "REGENERATE", "evidence": "can be re-executed", + "scope": "shared", "released": False, "rule": "r1"}, {"task": "run/c", "process": "JOIN", "target": "", "contribution": "IRREDUCIBLE", "storage": "WRITABLE", - "action": "REGENERATE", "reason": "no script recorded", - "exclusive": True, "terminal": False, "rule": "r2"}, + "action": "REGENERATE", "evidence": "no script recorded", + "scope": "exclusive", "released": False, "rule": "r2"}, ], } diff --git a/tests/test_runs.py b/tests/test_runs.py index 0bd1b8b..33d59b6 100644 --- a/tests/test_runs.py +++ b/tests/test_runs.py @@ -15,7 +15,6 @@ from clew.extract import runs -FIXTURES = Path(__file__).resolve().parent / "fixtures" def graph(name): @@ -145,54 +144,6 @@ def test_a_sidecar_filed_under_a_run_hash_is_still_read(self): self.assertEqual(g["output_details"]["aa/1"][0]["digest"], "sha256:old") self.assertEqual(g["published"]["p.txt"]["digest"], "sha256:old") -class LineageStoreRuns(unittest.TestCase): - """A session-id prefix names a resume chain; its newest run stands for it.""" - - def setUp(self): - from tests.test_lineage_store import RUN_A, RUN_B, RUN_C, CHAIN, OTHER - self.root = Path(tempfile.mkdtemp()) - history = self.root / ".history" - history.mkdir() - (history / RUN_A).write_text(f"2026-08-01 10:00:00 CEST\tfirst_run\t{CHAIN}\tlid://{RUN_A}\n") - (history / RUN_B).write_text(f"2026-08-02 10:00:00 CEST\tsecond_run\t{CHAIN}\tlid://{RUN_B}\n") - (history / RUN_C).write_text(f"2026-08-03 10:00:00 CEST\tother_run\t{OTHER}\tlid://{RUN_C}\n") - self.run_b = RUN_B - - def tearDown(self): - shutil.rmtree(self.root) - - def test_a_session_prefix_resolves_to_the_chain_s_newest_run(self): - store = runs.Runs(self.root) - self.assertEqual(store.resolve("session-ch"), ("second_run", self.run_b)) - self.assertEqual(store.resolve("bbbb"), ("second_run", self.run_b)) - - def test_a_prefix_spanning_two_sessions_is_still_ambiguous(self): - with self.assertRaises(SystemExit): - runs.Runs(self.root).resolve("session-") - - -class HorusRuns(unittest.TestCase): - def test_a_single_run_directory(self): - store = runs.Runs(FIXTURES / "horus_run") - self.assertEqual(store.kind, "horus-run") - self.assertEqual(store.records()[0]["timestamp"], "2026-09-02T09:01:34.050766+00:00") - g = store.load() - self.assertEqual(g["run"]["name"], "horus_run") - self.assertTrue(any(d.get("digest", "").startswith("sha256:") - for ds in g["output_details"].values() for d in ds)) - - def test_a_root_of_run_directories(self): - root = Path(tempfile.mkdtemp()) - try: - shutil.copytree(FIXTURES / "horus_run", root / "run-1") - shutil.copytree(FIXTURES / "horus_run", root / "run-2") - store = runs.Runs(root) - self.assertEqual(store.kind, "horus") - self.assertEqual(sorted(n for n, _, _ in store.names()), ["run-1", "run-2"]) - self.assertEqual(store.sidecar_path("run-1"), root / ".clew" / "run-1.digests.json") - finally: - shutil.rmtree(root) - if __name__ == "__main__": unittest.main() diff --git a/tests/test_workpath.py b/tests/test_workpath.py index dac09f4..a27a210 100644 --- a/tests/test_workpath.py +++ b/tests/test_workpath.py @@ -106,7 +106,7 @@ def test_classify_never_reports_destroyed_for_an_unplaced_task(self): facts = contribution.classify(graph, "trim/trimmed/s1.fq", exclusive=False, work_root=str(self.root)) self.assertIsNone(facts["storage"]) - self.assertIn("not placed under --work-root", facts["reason"]) + self.assertIn("not placed under --work-root", facts["evidence"]) def test_reclaim_keeps_every_task_of_a_shared_directory(self): # The Snakemake case: one directory for the whole workflow. A verdict