From 13433e44040ce037c1e6480f5c43560b30e841cc Mon Sep 17 00:00:00 2001 From: Hanqing Liu Date: Sat, 22 Aug 2026 14:59:30 -0400 Subject: [PATCH 1/2] the whitelist lands under the sample that reads it, so no planned job owns a path a second instance deletes The config records the registry name the KB chose and the consuming module builds the path, which is what lets that path be one no two jobs share: one owner for the location, so config and rule cannot disagree. `rule onlist` keeps temp(), keeps its verb, keeps no container, and gains a sample scope in both modules. Co-Authored-By: Claude Opus 5 (1M context) --- src/seqforge/compose/core.py | 58 ++++++++++++++++--------- src/seqforge/compose/params.py | 18 ++++---- src/seqforge/workflows/__init__.py | 20 ++++++++- src/seqforge/workflows/map/chromap.smk | 32 ++++++++------ src/seqforge/workflows/map/starsolo.smk | 31 ++++++++++--- 5 files changed, 108 insertions(+), 51 deletions(-) diff --git a/src/seqforge/compose/core.py b/src/seqforge/compose/core.py index d596f55..f9eaedb 100644 --- a/src/seqforge/compose/core.py +++ b/src/seqforge/compose/core.py @@ -157,8 +157,10 @@ class ComposePlan: units: list[dict[str, str]] module: ModuleSelection spec: Spec - #: relative path -> the registry name `rule onlist` will build it from. NOT the barcodes: - #: see `_resolve_token` and workflows/map/starsolo.smk for why compose does not write 111 MB. + #: The whitelists this recipe chose: backend param key -> the registry names its value renders, + #: in the value's own order (CB-position order for `soloCBwhitelist`). NOT the barcodes and NOT + #: paths — see `_resolve_token` for why compose writes neither 111 MB nor a location, and + #: workflows/map/starsolo.smk for where the consuming rule puts it. onlist_files: dict[str, list[str]] #: What the loaded spec's read floor admitted, or `None` when it declares none. The verdict lives #: here and never in the manifest: it is recomputed under whatever KB is loaded at compile time, @@ -612,28 +614,44 @@ def _resolve_params( registry: OnlistRegistry, onlist_files: dict[str, list[str]], ) -> dict[str, object]: - """Render the KB backend params for a CLI, resolving every ``{onlist:alias}`` to a real path.""" + """Render the KB backend params for a CLI, resolving every ``{onlist:alias}`` to a registry name.""" out: dict[str, object] = {} for key, value in spec.require_backend().params.items(): - if isinstance(value, list): - rendered = [_resolve_token(v, spec, registry, onlist_files) for v in value] - out[key] = " ".join(str(r) for r in rendered) - else: - out[key] = render_param(_resolve_token(value, spec, registry, onlist_files)) + values = value if isinstance(value, list) else [value] + rendered = [_resolve_token(v, spec, registry) for v in values] + out[key] = ( + " ".join(str(r) for r in rendered) + if isinstance(value, list) + else render_param(rendered[0]) + ) + # Recorded per KEY and in the value's own order, which for ``soloCBwhitelist`` is CB-position + # order: the whitelists this recipe CHOSE, for the gate, the tests and the instrument that + # materializes one for a run of its own. Deliberately not a path — the consuming rule builds + # a path scoped to the sample reading it, and a second spelling here is a second spelling + # free to disagree with the one the pipeline actually uses. + chosen = [ + str(r) for v, r in zip(values, rendered, strict=True) if _onlist_alias(v) is not None + ] + if chosen: + onlist_files[key] = chosen return out -def _resolve_token( - value: object, spec: Spec, registry: OnlistRegistry, onlist_files: dict[str, list[str]] -) -> object: - if not (isinstance(value, str) and value.startswith("{onlist:") and value.endswith("}")): +def _onlist_alias(value: object) -> str | None: + """The alias inside an ``{onlist:}`` token; ``None`` for any other backend param value.""" + if isinstance(value, str) and value.startswith("{onlist:") and value.endswith("}"): + return value[len("{onlist:") : -1] + return None + + +def _resolve_token(value: object, spec: Spec, registry: OnlistRegistry) -> object: + alias = _onlist_alias(value) + if alias is None: return value - alias = value[len("{onlist:") : -1] ref = spec.onlists.get(alias) if ref is None: raise ComposeError(f"backend references undeclared onlist alias {alias!r}") name = ref.registry - rel = f"onlists/{name}.txt" try: registry.packed(name) except OnlistNotAvailable as exc: @@ -641,12 +659,12 @@ def _resolve_token( f"onlist {name!r} is not materialized, so --soloCBwhitelist cannot be emitted: {exc}. " "Register it (URL + sha256) or run with a registry that can fetch it." ) from exc - # Verified, not materialized. `registry.packed(name)` above is what proves the whitelist EXISTS - # and is the declared barcode set -- refusing here, at compile time, is the rung-3 promise - # `compose` makes. What it must not do is write the 111 MB: that is `rule onlist`'s job, and - # `temp()`'s. The name is recorded so the gate and the tests can still see which list was chosen. - onlist_files[rel] = [name] - return rel + # Verified, not materialized, and what is emitted is the NAME. `registry.packed(name)` above is + # what proves the whitelist EXISTS and is the declared barcode set -- refusing here, at compile + # time, is the rung-3 promise `compose` makes. What it must not do is write the 111 MB, nor say + # where the file goes: building it is `rule onlist`'s job and the location is the consuming + # module's, which is what lets that location be one no two jobs share. + return name def _read_files_in(manifest: DatasetManifest, module: WorkflowModule) -> dict[str, str]: diff --git a/src/seqforge/compose/params.py b/src/seqforge/compose/params.py index 1eaea66..21f7218 100644 --- a/src/seqforge/compose/params.py +++ b/src/seqforge/compose/params.py @@ -439,14 +439,14 @@ def render_param(value: object) -> str: return str(value) -def _resolves_to_onlist_path(value: object) -> bool: +def _resolves_to_onlist_name(value: object) -> bool: """A KB param whose value is an ``{onlist:}`` token, or a list of them. - Such a value is resolved to a materialized whitelist PATH at compose time (see - ``compose.core._resolve_token``), so its config rendering is a path, not the verbatim token — the - per-key faithfulness check must skip it or it would compare a path against a token and always fail. - Both STARsolo's ``soloCBwhitelist`` and chromap's ``barcode_whitelist`` are such params; keying on - the VALUE rather than the key name covers a third one without spelling it out. + Such a value is resolved to the whitelist's REGISTRY NAME at compose time (see + ``compose.core._resolve_token``), so its config rendering is that name, not the verbatim token — + the per-key faithfulness check must skip it or it would compare a name against an alias and + always fail. Both STARsolo's ``soloCBwhitelist`` and chromap's ``barcode_whitelist`` are such + params; keying on the VALUE rather than the key name covers a third one without spelling it out. """ values = value if isinstance(value, list) else [value] return any(isinstance(v, str) and v.startswith("{onlist:") for v in values) @@ -536,9 +536,9 @@ def params_gate( # ---- 3. faithfulness, per key, per owner ---- for key, expected in params.items(): - if _resolves_to_onlist_path(expected): - continue # an {onlist:...} token is resolved to a path at compose, so it is not - # compared verbatim (the registry proves the whitelist exists in `_resolve_token`). + if _resolves_to_onlist_name(expected): + continue # an {onlist:...} token is resolved to a registry name at compose, so it is + # not compared verbatim (the registry proves the whitelist exists in `_resolve_token`). # Value-based, not a key name: covers STARsolo's `soloCBwhitelist` AND chromap's # `barcode_whitelist` without either being spelled out here. want = render_param(expected) diff --git a/src/seqforge/workflows/__init__.py b/src/seqforge/workflows/__init__.py index 176b84c..22a1e88 100644 --- a/src/seqforge/workflows/__init__.py +++ b/src/seqforge/workflows/__init__.py @@ -35,6 +35,22 @@ from ..kb.schema import Spec #: CalVer YYYY.M.PATCH; bump when any shipped module's rules/params change. +#: 2026.8.25 — `rule onlist` materializes a whitelist under the SAMPLE that reads it, in `map/starsolo` +#: and `map/chromap` (#488, the last case of #478's class). It owned `onlists/.txt` — one ~115 MB +#: file every sample in the deposit maps against — and snakemake removes an output before running the +#: job that makes it, so a second instance over one results directory raced the first's REMOVAL: the +#: hazard measured at 0 of 6 on the index rule, where an atomic idempotent body still gave 0 of 6. +#: The rule stays local, stays without a `container:`, stays the `seqforge io onlist write` verb, and +#: stays `temp()`; only its output gains a sample scope. **What moves is a location, never a byte**: +#: the aligner reads the same barcodes and every count, matrix and record is identical. The emitted +#: config now records the whitelist's registry NAME rather than a path, and the consuming module +#: builds the path from it — one owner for the location, so config and rule cannot disagree — which +#: is what makes the sample scope expressible at all, since a path fixed at compile time is one no +#: job can vary. A chemistry declaring three whitelists still renders three paths in CB-position +#: order into one argv token. `temp()` now reclaims each copy after the job that read it, so the peak +#: is one copy per aligner job in flight (three for those chemistries) rather than one per compiled +#: run kept forever — which is not the permanent duplication ADR-0015 rejected. `required_config` is +#: unchanged for both. #: 2026.8.24 — `rule umi_count` says how much of each cell's yield was ribosomal (#471). Two more #: `obs` columns on the plate object — `n_umis_rrna` and `rrna_fraction`, both counted over #: `umi_combined`, the same population `n_umis` and the saturation are read off — and the page carries @@ -623,7 +639,7 @@ #: dereferenced and never declared. The contract was wrong, not the module. #: 2026.7.1 — star.smk hardcodes --outSAMtype (it is a module detail, and starsolo.smk always #: hardcoded it); required_config gains primary_feature and drops bulk.outSAMtype. -WORKFLOW_VERSION = "2026.8.24" +WORKFLOW_VERSION = "2026.8.25" _MODULE_DIR = Path(__file__).parent @@ -792,7 +808,7 @@ def argv_keys_read_by(source: Path) -> frozenset[str]: #: chromap's parse namespace — the byte-decided knobs a ``map/chromap`` backend may declare. Just the #: barcode whitelist: chromap corrects the cell barcode against it (like STARsolo's ``soloCBwhitelist``), -#: and it resolves through the same ``{onlist:}`` mechanism to a materialized path. Everything +#: and it resolves through the same ``{onlist:}`` mechanism to a registry name. Everything #: else chromap needs is either a fixed module detail (the ``--preset atac`` mode, hardcoded in #: chromap.smk the way star.smk hardcodes ``--outSAMtype``) or read geometry the manifest already states #: (which file is the barcode read arrives via ``read_files_in``, not a parse param). The namespace is diff --git a/src/seqforge/workflows/map/chromap.smk b/src/seqforge/workflows/map/chromap.smk index fd3e04e..46242ad 100644 --- a/src/seqforge/workflows/map/chromap.smk +++ b/src/seqforge/workflows/map/chromap.smk @@ -52,14 +52,17 @@ def commajoin(sample, role): return ",".join(fastqs(sample, role)) -def whitelist(): - """The materialized barcode whitelist path — the config value is the argv rendering. - - `config["chromap"]["barcode_whitelist"]` is `onlists/.txt`, built on demand by `rule onlist` - and `temp()`-deleted after — the same packed-onlist discipline starsolo.smk uses for - `soloCBwhitelist`, so a 111 MB whitelist is never written into the run directory at compile time. +def whitelist(sample): + """This SAMPLE's barcode whitelist — the config names the list, this module owns the path. + + `config["chromap"]["barcode_whitelist"]` is the registry NAME the KB chose; the file is built on + demand by `rule onlist` under a directory named for the sample that reads it and `temp()`-deleted + after — the same packed-onlist discipline starsolo.smk uses for `soloCBwhitelist`, so a 111 MB + whitelist is never written into the run directory at compile time and no job here owns a path a + second snakemake instance over one results directory would delete out from under the first. """ - return CHROMAP["barcode_whitelist"] + name = CHROMAP["barcode_whitelist"] + return f"onlists/{sample}/{name}.txt" @cache @@ -117,15 +120,18 @@ rule all: rule onlist: - """Materialize one barcode whitelist for chromap to read once and snakemake to then delete. + """Materialize one sample's barcode whitelist for chromap to read once and snakemake to delete. Byte-for-byte the same rule as starsolo.smk's — a barcode whitelist is a barcode whitelist. `temp()` - is the point: the ARC list is materialized on demand and deleted when the last job needing it is - done, never written into the run directory at compile time. No `container:`: this runs `seqforge`, - not an aligner, so the ambient compose environment already has it. + is the point: the ARC list is materialized on demand and deleted when the job needing it is done, + never written into the run directory at compile time. The SAMPLE in the path is the second half: + one file the whole deposit shares is a path every concurrent snakemake instance over one results + directory owns, and snakemake deletes an output before running the job that produces it — so the + second instance races the first's removal, which is 0 of 6 surviving and not a slow write. No + `container:`: this runs `seqforge`, not an aligner, so the ambient compose environment has it. """ output: - temp("onlists/{name}.txt"), + temp("onlists/{sample}/{name}.txt"), localrule: True shell: "seqforge io onlist write {wildcards.name} --out {output}" @@ -143,7 +149,7 @@ rule chromap_align: gdna1=lambda wc: fastqs(wc.sample, config["read_files_in"]["gdna1"]), gdna2=lambda wc: fastqs(wc.sample, config["read_files_in"]["gdna2"]), barcode=lambda wc: fastqs(wc.sample, config["read_files_in"]["barcode"]), - whitelist=whitelist(), + whitelist=lambda wc: whitelist(wc.sample), output: fragments=temp(f"{OUTDIR}/{{sample}}/{RAW_FRAGMENTS}"), # The pinned aligner: liulab-runtime's `align-dna`, resolved by compose to a ghcr tag or a prebuilt diff --git a/src/seqforge/workflows/map/starsolo.smk b/src/seqforge/workflows/map/starsolo.smk index e4292d9..b408715 100644 --- a/src/seqforge/workflows/map/starsolo.smk +++ b/src/seqforge/workflows/map/starsolo.smk @@ -93,9 +93,21 @@ def readfilesin(sample, *roles): return " ".join(",".join(fastqs(sample, role)) for role in roles) -def whitelists(): - """One path for 10x; three for a split-pool chemistry. The config value is the argv rendering.""" - return SOLO["soloCBwhitelist"].split() +def whitelists(sample): + """This SAMPLE's barcode whitelists -- one for 10x, three for a split-pool chemistry, in order. + + The config names the whitelists the KB chose (registry names, CB-position order, one argv token); + the PATH is this module's, and it is scoped to the sample that reads it. That split is what keeps + the two from disagreeing -- there is no location in the config to drift from the location the rule + declares -- and it is why no job here owns a path a second snakemake instance over one results + directory would delete out from under the first. A whitelist is ~115 MB and every sample in a + deposit maps against the same one, so it was the last such path: `temp()` now reclaims each copy + when the job that read it is done, which costs one copy per aligner job in flight (three for the + chemistries declaring three) against a STAR job of tens of minutes, rather than a copy per + compiled run kept forever. The order is the config's, unchanged: STARsolo pairs the Nth whitelist + with the Nth CB position. + """ + return [f"onlists/{sample}/{name}.txt" for name in SOLO["soloCBwhitelist"].split()] @cache @@ -172,18 +184,23 @@ rule all: rule onlist: - """Materialize one barcode whitelist, for STAR to read once and snakemake to then delete. + """Materialize one sample's barcode whitelist, for STAR to read once and snakemake to then delete. `temp()` is the entire point, and why -- the 111 MB a compiled run used to carry three times over, and why an input with no producing rule was `temp()`-able in name only -- is argued once in - ADR-0015. + ADR-0015. The SAMPLE in the path is the other half and is not decoration: this output used to be + one file the whole deposit shared, and snakemake deletes an output before running the job that + produces it, so a second instance over one results directory raced the first's REMOVAL rather + than its write -- measured on the index rule as 0 of 6 instances surviving, and an atomic, + idempotent body still 0 of 6, because the window is not the rule's to close. A copy nobody else + can claim leaves nothing to race. See `whitelists` for what it costs. No `container:` directive, deliberately. This runs `seqforge`, which is not an aligner -- the ambient environment is the one that just ran `seqforge compose`, so it is by construction the one that has it. Naming `align-rna` here would put our own tool inside STAR's image. """ output: - temp("onlists/{name}.txt"), + temp("onlists/{sample}/{name}.txt"), localrule: True shell: "seqforge io onlist write {wildcards.name} --out {output}" @@ -298,7 +315,7 @@ rule starsolo_count: cdna=lambda wc: fastqs(wc.sample, config["read_files_in"]["cdna"]), barcode=lambda wc: fastqs(wc.sample, config["read_files_in"]["barcode"]), loaded=rules.load_genome.output, - whitelist=whitelists(), + whitelist=lambda wc: whitelists(wc.sample), output: # `temp()` on everything: the raw matrices are consumed by `solo_to_h5ad`, the stats + # filtered tree + logs by `qc_bundle`, and the BAM by `solo_to_cram`. Snakemake deletes each From 89e82a4548a4433c9c5e770d5224aa9d42b4cb64 Mon Sep 17 00:00:00 2001 From: Hanqing Liu Date: Sat, 22 Aug 2026 14:59:46 -0400 Subject: [PATCH 2/2] the shared-path invariant holds the whitelist to it, and the plan says where the copy lands and when it goes RULES_THAT_MAY_OWN_A_SHARED_PATH is down to the load flag, the one path measured safe to race. The two claims a module's source text used to carry -- temp() and the producing verb -- move onto the rendered plan, where a rename cannot break them falsely and an indirection cannot pass them falsely, and the split-pool dry run gains the three paths in CB-position order as one argv token. Co-Authored-By: Claude Opus 5 (1M context) --- tests/test_compose.py | 73 ++++++++++++++++++++++++----------- tests/test_repo_invariants.py | 2 +- tests/test_workflows.py | 20 +++++----- 3 files changed, 61 insertions(+), 34 deletions(-) diff --git a/tests/test_compose.py b/tests/test_compose.py index fecb391..054486f 100644 --- a/tests/test_compose.py +++ b/tests/test_compose.py @@ -31,7 +31,6 @@ _processing, _rendered_shell, _rule_blocks, - _src_root, count_matrix, declare_read_floor, one_run_each, @@ -83,20 +82,17 @@ def test_compose_10x_emits_kb_params_and_passes_the_params_gate( # read the path compose REPORTS, not one reconstructed here: the layout is keyed by run_id and a # test that hardcodes it is testing its own arithmetic - pipeline_dir = (tmp_path / result.config_path).parent config = yaml.safe_load((tmp_path / result.config_path).read_text()) assert config["solo"]["soloCBlen"] == "16" assert config["solo"]["soloUMIlen"] == "12" assert config["solo"]["soloStrand"] == "Forward" # --readFilesIn order: the cDNA read precedes the barcode read assert config["read_files_in"] == {"cdna": "R2", "barcode": "R1"} - # The whitelist token resolved to a PATH, and compose did not write the file. `rule onlist` - # builds it and `temp()` deletes it: 10x's real v3 list is 111 MB of text, and writing it into - # every run directory at compile time cost a third of a gigabyte for one dataset compiled three - # ways -- for a file STAR opens once. Compose still VERIFIES the registry can produce it, which - # is the compile-time refusal that matters. - assert config["solo"]["soloCBwhitelist"] == "onlists/3M-february-2018.txt" - assert not (pipeline_dir / config["solo"]["soloCBwhitelist"]).exists() + # The whitelist token resolved to the registry NAME -- the choice, which is compose's, and not a + # location, which is the consuming rule's. Compose still VERIFIES the registry can produce the + # list, which is the compile-time refusal that matters; that it writes none of the 111 MB is + # `test_the_whitelist_is_a_rule_output_not_a_compile_time_write`'s claim. + assert config["solo"]["soloCBwhitelist"] == "3M-february-2018" units = (tmp_path / result.units_path).read_text().splitlines() assert units[0].split("\t") == ["sample_id", "run", "lane", "read_id", "path"] @@ -162,6 +158,19 @@ def test_compose_bd_enhanced_derives_the_adapter_anchored_starsolo_recipe( ) assert f"--soloAdapterSequence {solo['soloAdapterSequence']}" in planned assert f"--soloCBposition {solo['soloCBposition']}" in planned + # THREE whitelists, as three sample-scoped paths in ONE argv token, in the order the config + # names them — which is CB-position order, and STARsolo pairs the Nth list with the Nth position. + # The config names the lists and this module builds the paths (#488), so the ordering claim the + # config-level sweep makes is only half of one: a module free to build the paths is free to sort + # them, and a mis-paired round produces a plausible matrix at exit 0. Both halves are asserted + # against the same rendered command a split-pool sample would really be counted with. + sample = manifest.experiment.samples[0].sample_id + lists = " ".join(f"onlists/{sample}/{n}.txt" for n in str(solo["soloCBwhitelist"]).split()) + assert len(lists.split()) == 3, f"bd-enhanced no longer carries three whitelists: {lists}" + assert f"--soloCBwhitelist {lists} " in planned, ( + f"the three whitelists do not reach STAR as one token in CB-position order. " + f"Planned:\n{planned}" + ) # The Complex half of the barcode-match mode, in argv, where it is decidable. `1MM` was a literal # in `cb_umi_geometry()`'s Complex branch until #198; it is now the KB's, and this asserts BOTH # halves of that move — it arrives (a Complex chemistry that named no match type would FATAL on @@ -493,10 +502,8 @@ def test_the_wiring_gate_fails_a_workflow_that_plans_nothing( run_dir.mkdir() (run_dir / "config.yaml").write_text(yaml.safe_dump(p.config, sort_keys=True)) (run_dir / "units.tsv").write_text(core._units_tsv(p.units)) - for rel, lines in p.onlist_files.items(): - t = run_dir / rel - t.parent.mkdir(parents=True, exist_ok=True) - t.write_text("\n".join(lines) + "\n") + # No whitelist is staged, and none is owed: it is a rule output with a producing rule, so a plan + # that reaches it plans that rule, and this plan reaches nothing at all — which is the subject. module = get_module(p.module.name) # the OLD wrapper, verbatim in shape: include: instead of module/use rule (run_dir / "Snakefile").write_text( @@ -525,10 +532,13 @@ def test_the_composed_pipeline_plans_the_h5ad_the_whitelist_and_the_command_star give. "One file per sample, and it waits for the sample" is `test_rule_all_names_one_file_per_sample_and_that_file_waits_for_the_whole_sample`'s claim over every module; what this owns is that the object survived the target list shrinking. - 2. **The whitelist has a producing job and is temporary.** A rule declared above `rule all` - becomes the workflow's default target, and a default target with a wildcard is a hard snakemake - error — which is exactly what the first attempt at `rule onlist` did. A dry run is the only - thing that knows. + 2. **The whitelist has a producing job, lands under the sample that reads it, and is reclaimed.** + A rule declared above `rule all` becomes the workflow's default target, and a default target + with a wildcard is a hard snakemake error — which is exactly what the first attempt at + `rule onlist` did. A dry run is the only thing that knows. The sample scope is the same claim + `test_no_planned_job_owns_a_path_a_concurrent_instance_would_delete` makes for every module; + what is here is the exact path and its reclamation line, which is what says the location was + not bought by leaving 115 MB of barcodes in the run directory (#488). 3. **`--soloBarcodeReadLength 0` reaches STAR.** The module reads it with `SOLO.get(...)`, not a subscript, so that it stays optional for chemistries that do not declare it; nothing but the argv proves the 10x half of that still arrives. @@ -571,7 +581,23 @@ def test_the_composed_pipeline_plans_the_h5ad_the_whitelist_and_the_command_star assert f"{sample}.velocyto.h5ad" in planned assert "rule onlist" in planned, "the whitelist has no producing job in the plan" - assert "3M-february-2018" in planned + assert "seqforge io onlist write 3M-february-2018" in planned, ( + f"the producing job does not build the chosen list with the verb. Planned:\n{planned}" + ) + # ...under THIS SAMPLE, and reclaimed after the job that read it. One file every sample of a + # deposit shares is a path every concurrent snakemake instance over one results directory owns, + # and snakemake removes an output before running the job that makes it — measured on the index + # rule as 0 of 6 instances surviving, an atomic idempotent body still 0 of 6. The reclamation + # line is the other half: the location may not be bought by leaving 115 MB in the run directory. + # Both are read off the plan rather than the module's source text, which a rename breaks falsely + # and an indirection passes falsely. + whitelist = f"onlists/{sample}/3M-february-2018.txt" + assert f"output: {whitelist}" in planned, ( + f"the whitelist is not planned under the sample that reads it. Planned:\n{planned}" + ) + assert f"Would remove temporary output {whitelist}" in planned, ( + f"the whitelist copy is never reclaimed. Planned:\n{planned}" + ) assert "--soloBarcodeReadLength 0" in planned @@ -3030,6 +3056,13 @@ def test_the_whitelist_is_a_rule_output_not_a_compile_time_write( `temp()` was also meaningless before this rule existed: the whitelist was bound to `starsolo_count.input` with no producing rule, and snakemake cannot delete a file it did not make. + + What compose does NOT write is the whole of this test, and it is the half no plan can show. The + other half — that a rule produces the file, under the sample that reads it, and that the plan + reclaims it afterwards — is read off the rendered plan by + `test_the_composed_pipeline_plans_the_h5ad_the_whitelist_and_the_command_star_receives`, where it + costs no second spawn. It used to be asserted here against the shipped module's SOURCE TEXT, + which a rename breaks falsely and an indirection passes falsely. """ manifest, reg = built_v3 processing = _processing(manifest) @@ -3037,10 +3070,6 @@ def test_the_whitelist_is_a_rule_output_not_a_compile_time_write( pipeline_dir = (tmp_path / result.config_path).parent assert not (pipeline_dir / "onlists").exists(), "compose wrote the whitelist" - module = (_src_root() / "workflows" / "map" / "starsolo.smk").read_text() - assert 'temp("onlists/{name}.txt")' in module - assert "seqforge io onlist write" in module - # ---- the live KB's read floor drops a starved cell, and the manifest keeps every one (#292) ---- # diff --git a/tests/test_repo_invariants.py b/tests/test_repo_invariants.py index 88b6224..3fb47c3 100644 --- a/tests/test_repo_invariants.py +++ b/tests/test_repo_invariants.py @@ -904,7 +904,7 @@ def fires(source: str) -> bool: assert not fires( " # one gzipped .qc.json.gz per sample, then temp() drops the rest" ) # prose - assert not fires(' temp("onlists/{name}.txt"),') # a name no module publishes + assert not fires(' temp("onlists/{sample}/{name}.txt"),') # a name no module publishes _REPO = Path(__file__).resolve().parent.parent diff --git a/tests/test_workflows.py b/tests/test_workflows.py index 1e3a55f..c59cdf1 100644 --- a/tests/test_workflows.py +++ b/tests/test_workflows.py @@ -1698,15 +1698,13 @@ def test_rule_all_names_one_file_per_sample_and_that_file_waits_for_the_whole_sa #: Rules that own a path the whole run shares and which this invariant still tolerates, with why. #: ``load_genome``'s is a ``touch()`` marker and is measured SAFE to race — six instances at once #: over one directory, five trials, zero errors — because re-touching a flag that is already there -#: says exactly what the flag says. ``onlist`` stages a chemistry's barcode whitelist, which is one -#: 111 MB file the whole deposit maps against: a per-sample copy is the wrong answer, no measurement -#: says a concurrent rewrite is safe, and it is untouched by the index change. It is NAMED rather -#: than quietly matched because the two plate modules — where a sharded run is what anyone actually -#: does — declare no such rule, so the exposure and the fix belong to a ticket of their own. +#: says exactly what the flag says. It is the only one: the barcode whitelist was the last of the +#: others and is now materialized under the sample that reads it, so nothing is tolerated here that +#: was not measured. #: #: Named rather than skipped, in the shape :data:`MODULES_NO_SPEC_REACHES_YET` established, and the -#: case below asserts each name is still EARNING its place in the module that declares it. -RULES_THAT_MAY_OWN_A_SHARED_PATH: frozenset[str] = frozenset({"load_genome", "onlist"}) +#: case below asserts the name is still EARNING its place in the module that declares it. +RULES_THAT_MAY_OWN_A_SHARED_PATH: frozenset[str] = frozenset({"load_genome"}) @pytest.mark.parametrize("module", list_modules()) @@ -1740,10 +1738,10 @@ def test_no_planned_job_owns_a_path_a_concurrent_instance_would_delete( whether two instances collide is whether the PATH differs, and a per-sample rule writing to one fixed name collides exactly as the index rule did. - Parametrised over every registered module, so a sixth cannot land without an answer. The two - rules whose shared path is tolerated are :data:`RULES_THAT_MAY_OWN_A_SHARED_PATH`, with the - reason each is there; a shared output added to one of THOSE is what this cannot see, so the - load flag is additionally held to being a single file. + Parametrised over every registered module, so a sixth cannot land without an answer. The one + rule whose shared path is tolerated is :data:`RULES_THAT_MAY_OWN_A_SHARED_PATH`, with the reason + it is there; a shared output added to THAT one is what this cannot see, so the load flag is + additionally held to being a single file. """ tech, assembly = planning_route(module) manifest, reg = _build(tmp_path, tech)